diff --git a/.gitattributes b/.gitattributes index 014da27ce00016e8eb796b31d4cb97bfaad21337..754d14b751359813c2570fdd43ca50c7f5cd6cda 100644 --- a/.gitattributes +++ b/.gitattributes @@ -358,3 +358,20 @@ aft_wave_v2/charter_real_4x__agreement/training/checkpoints/checkpoint-512/token aft_wave_v2/charter_real_4x__agreement/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_wave_v2/charter_real_4x__agreement/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_wave_v2/charter_real_4x__agreement/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/ARTIFACT_MANIFEST.local.json b/aft_wave_v2/coin_real_4x__charter0p2/training/ARTIFACT_MANIFEST.local.json new file mode 100644 index 0000000000000000000000000000000000000000..4fd9d1421e9e4789078dfac767fc5e5290010b5a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/ARTIFACT_MANIFEST.local.json @@ -0,0 +1,867 @@ +{ + "repo": "arcadia-impact/scimt-dispatch-models", + "remote_prefix": "aft_wave_v2/coin_real_4x__charter0p2/training", + "local_folder": "/workspace/wave/training", + "files": { + "TRAINED.json": { + "size": 965, + "sha256": "262e91d72c2c2763dfe35d44f2ee3c101b7f5b6185495f379c600571880b390b" + }, + "axolotl.yaml": { + "size": 1212, + "sha256": "bc4c1b11c8ee0a514f0c2c3ee14ee86d53a61bcc6e24e4bd12e351560b4cac09" + }, + "checkpoint.json": { + "size": 2232, + "sha256": "01c471219eaf40c78ce3649da407f290e32fb8a21e54e0026c207052325b8450" + }, + "checkpoints/README.md": { + "size": 2940, + "sha256": "71baf553b2923beebe99d9397143fe7aea49d5c041b6b6916f634cc69ba6e1a6" + }, + "checkpoints/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/adapter_model.safetensors": { + "size": 547777976, + "sha256": "777863e8bd407287f6fc4ae90c64cd69b9931e7cafe53c7e9a27afc677d964a1" + }, + "checkpoints/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-128/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-128/adapter_model.safetensors": { + "size": 547777976, + "sha256": "967756a76d766634a937fc30d741739361630793875f04d2afc07f8b79a88139" + }, + "checkpoints/checkpoint-128/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/optimizer.pt": { + "size": 1048106435, + "sha256": "37b4c7084d4d2fec32863d93caa58aceb115bc383fdf6a785da612b00f880e36" + }, + "checkpoints/checkpoint-128/rng_state.pth": { + "size": 14645, + "sha256": "234e8c0c3fd2b805b594ff492259b2db7aead6c3ca1abbea3168b40935e4d0cb" + }, + "checkpoints/checkpoint-128/scheduler.pt": { + "size": 1465, + "sha256": "efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f" + }, + "checkpoints/checkpoint-128/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-128/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-128/tokens_state.json": { + "size": 38, + "sha256": "626a3a668af45684d29aa868c59a31a375f48326b3e3477ec7a45f305e4e9263" + }, + "checkpoints/checkpoint-128/trainer_state.json": { + "size": 56231, + "sha256": "69d6aea8cdfef13dff2547995ece9647521e93db40e1575d156e65fe5028d677" + }, + "checkpoints/checkpoint-128/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-160/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-160/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-160/adapter_model.safetensors": { + "size": 547777976, + "sha256": "c0af7b1e609afc4f1956162573e14100202cddc064e6c56d0d06f38d83cf5f79" + }, + "checkpoints/checkpoint-160/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-160/optimizer.pt": { + "size": 1048106435, + "sha256": "10669e844f02a2ed2fb906b3556f11a110eb11b33abfd347d76b78563ccb11b4" + }, + "checkpoints/checkpoint-160/rng_state.pth": { + "size": 14645, + "sha256": "62012b9e748bbef1beb75187c13d345f2d456800a4ec5207b243fb40c95ada94" + }, + "checkpoints/checkpoint-160/scheduler.pt": { + "size": 1465, + "sha256": "69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89" + }, + "checkpoints/checkpoint-160/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-160/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-160/tokens_state.json": { + "size": 38, + "sha256": "52e1c49d14095bc8dac2d7ce56d7a7d2565d1eaafd9d642f6ec87983f564a61e" + }, + "checkpoints/checkpoint-160/trainer_state.json": { + "size": 70211, + "sha256": "ce09d7f73968164fa7cd219c8b128a83830df9dea11d442711813e89a6cae3e3" + }, + "checkpoints/checkpoint-160/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-192/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-192/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-192/adapter_model.safetensors": { + "size": 547777976, + "sha256": "bb2d6e80c9b66d62b36ae40e54bbbcdf0c97db5c46489f1828b7a33386a0cd71" + }, + "checkpoints/checkpoint-192/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-192/optimizer.pt": { + "size": 1048106435, + "sha256": "669b3fb6f447d4c422509832d768e3d5b970e63da72ded60c0a929f4ae9cb1bf" + }, + "checkpoints/checkpoint-192/rng_state.pth": { + "size": 14645, + "sha256": "3d69b2462378ed67f5f5555fb6b192965044d7652528913ba5ccc13c8e15172a" + }, + "checkpoints/checkpoint-192/scheduler.pt": { + "size": 1465, + "sha256": "076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd" + }, + "checkpoints/checkpoint-192/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-192/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-192/tokens_state.json": { + "size": 38, + "sha256": "56653a01ea2a528ef30b635baa652e642b815967a6febe58b4d5c0bf26ea3590" + }, + "checkpoints/checkpoint-192/trainer_state.json": { + "size": 84198, + "sha256": "8152227e4d02dac10ab9ac21413688fcf5b48acd60bc66b6f0e4c09e9fa66241" + }, + "checkpoints/checkpoint-192/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-224/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-224/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-224/adapter_model.safetensors": { + "size": 547777976, + "sha256": "6888ee85bd7aa03e144cb6712ed48bb2f6c946fd988821e8c1d171f0534cfdac" + }, + "checkpoints/checkpoint-224/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-224/optimizer.pt": { + "size": 1048106435, + "sha256": "c816093831336bcd8ff55e56bcbfd2c9c8c2cf76714d1bdae3d38dac7225263f" + }, + "checkpoints/checkpoint-224/rng_state.pth": { + "size": 14645, + "sha256": "496ba904d564744600c2dc9631c09eb565ed0719573f403cd75b1e83021c01e1" + }, + "checkpoints/checkpoint-224/scheduler.pt": { + "size": 1465, + "sha256": "f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822" + }, + "checkpoints/checkpoint-224/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-224/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-224/tokens_state.json": { + "size": 39, + "sha256": "16868a422e40c264a6ca0bde115aac3a878f4138b3eb5b6cbdb5317d3076c77b" + }, + "checkpoints/checkpoint-224/trainer_state.json": { + "size": 98198, + "sha256": "5934cccdf4465517fb839f4ce73be46c4449b87a3d954aceb05487641a0d6cb3" + }, + "checkpoints/checkpoint-224/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-256/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-256/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-256/adapter_model.safetensors": { + "size": 547777976, + "sha256": "8126745e8b4964386549b1d74f44cddc098688536a282ea0ab2078cae3e34684" + }, + "checkpoints/checkpoint-256/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-256/optimizer.pt": { + "size": 1048106435, + "sha256": "d5ae1fcce8ae0ed405b5ead388aa23f13b15ac16b0caa86c5be38cddc383e8d8" + }, + "checkpoints/checkpoint-256/rng_state.pth": { + "size": 14645, + "sha256": "370b4fba725358746db83895d8635e99b2f70cb36421eb17a0433851f5afd453" + }, + "checkpoints/checkpoint-256/scheduler.pt": { + "size": 1465, + "sha256": "1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3" + }, + "checkpoints/checkpoint-256/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-256/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-256/tokens_state.json": { + "size": 39, + "sha256": "a85de408b5cf3e4ec60a2f664d4ea579669320f6d816f5e149ddc498d54af418" + }, + "checkpoints/checkpoint-256/trainer_state.json": { + "size": 112255, + "sha256": "1437a6014c9feb6f6205361449c5426360b6af38149d736fd3bfdcf60b455006" + }, + "checkpoints/checkpoint-256/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-288/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-288/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-288/adapter_model.safetensors": { + "size": 547777976, + "sha256": "5a5fac124504ab1bc7f458cbe2dc24902a6eed60990525931d83e9b1e479c943" + }, + "checkpoints/checkpoint-288/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-288/optimizer.pt": { + "size": 1048106435, + "sha256": "c63d60019e5c7241ab66e80442d72ac275a130d591e250fe206fa8b7674585d4" + }, + "checkpoints/checkpoint-288/rng_state.pth": { + "size": 14645, + "sha256": "47ff3a93a2ca5e4e3e5cc87da30e0868111aa8accfb80c30aa2ec9467da212fb" + }, + "checkpoints/checkpoint-288/scheduler.pt": { + "size": 1465, + "sha256": "a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363" + }, + "checkpoints/checkpoint-288/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-288/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-288/tokens_state.json": { + "size": 39, + "sha256": "269f96db3710e3541266527aebd6cf43e7cd78f17ad6edf08dce39315676b4c2" + }, + "checkpoints/checkpoint-288/trainer_state.json": { + "size": 126343, + "sha256": "8d2b83396e7ffcb8587414ae1d0bb2e512617f08b3c094efae71a4b0dee9207b" + }, + "checkpoints/checkpoint-288/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-32/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-32/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-32/adapter_model.safetensors": { + "size": 547777976, + "sha256": "dfc7137dc4b591da53441e1cb4116357f7243c067b8999be573d6f5755f775f6" + }, + "checkpoints/checkpoint-32/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-32/optimizer.pt": { + "size": 1048106435, + "sha256": "29c1a2beb912c4b336bda6b7573f4831da7d1bdfb786616a94ae34fbd4be3e04" + }, + "checkpoints/checkpoint-32/rng_state.pth": { + "size": 14645, + "sha256": "046455cb6ded9b2989f35bf5efbbff0517d6651771b52c93d31759906cafe479" + }, + "checkpoints/checkpoint-32/scheduler.pt": { + "size": 1465, + "sha256": "8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc" + }, + "checkpoints/checkpoint-32/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-32/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-32/tokens_state.json": { + "size": 37, + "sha256": "cec49603be80ef1df2a4dadb25de3e8be8585000fa9c1e62970e67efa8f9d602" + }, + "checkpoints/checkpoint-32/trainer_state.json": { + "size": 14418, + "sha256": "2f2f0f0add7c8affe53373ba2b8578438902fb55ee4c555013a110d2b755056e" + }, + "checkpoints/checkpoint-32/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-320/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-320/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-320/adapter_model.safetensors": { + "size": 547777976, + "sha256": "e765cc322f7617de6546b93215e54b9d37ec3845fee1437a479e097997fb1ef8" + }, + "checkpoints/checkpoint-320/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-320/optimizer.pt": { + "size": 1048106435, + "sha256": "8cfb21d1144d02293b3bc081231fece171daac061c3242d61e3def1b9a061ee3" + }, + "checkpoints/checkpoint-320/rng_state.pth": { + "size": 14645, + "sha256": "0391605f7d7b8626c1c97d35086145a3e9e4df5cafc3842162a7664ce0f73f43" + }, + "checkpoints/checkpoint-320/scheduler.pt": { + "size": 1465, + "sha256": "e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0" + }, + "checkpoints/checkpoint-320/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-320/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-320/tokens_state.json": { + "size": 39, + "sha256": "5dba8649e21e024152b4b3daeba662208bb0f33796ac11478c9b0d7cffa65e97" + }, + "checkpoints/checkpoint-320/trainer_state.json": { + "size": 140448, + "sha256": "3b8b8ed1b05757ce1a069f6b317dec4e8dd3fe7e649fa7934310634d3c44fe33" + }, + "checkpoints/checkpoint-320/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-352/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-352/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-352/adapter_model.safetensors": { + "size": 547777976, + "sha256": "fd637c8c41ed471884c88ca6a4a3c769c09fe584c0234f1f3b6c4055a29661b9" + }, + "checkpoints/checkpoint-352/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-352/optimizer.pt": { + "size": 1048106435, + "sha256": "a9ae8f8539965fa44c77134ba73403bebdea09aa875274084a61cc75d423cdb1" + }, + "checkpoints/checkpoint-352/rng_state.pth": { + "size": 14645, + "sha256": "a5d3595bca569f2399926c6c523fd98aead95fc6d00353c61841e5028ebbdeff" + }, + "checkpoints/checkpoint-352/scheduler.pt": { + "size": 1465, + "sha256": "153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7" + }, + "checkpoints/checkpoint-352/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-352/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-352/tokens_state.json": { + "size": 40, + "sha256": "8b8f1db9f0311bacf95a403fdc167f8bce05fa9b5631e294453130a08f29b9c4" + }, + "checkpoints/checkpoint-352/trainer_state.json": { + "size": 154554, + "sha256": "483d66d68cb7524bcef0156d5584118b5f44c2770ef4159cc9b142f03cd48080" + }, + "checkpoints/checkpoint-352/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-384/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-384/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-384/adapter_model.safetensors": { + "size": 547777976, + "sha256": "cae387c1646783c110effda19caf750757601f7e073cbbb19fb98587cdd29ae7" + }, + "checkpoints/checkpoint-384/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-384/optimizer.pt": { + "size": 1048106435, + "sha256": "9bfd9862bf9544ac38b09870a31d1d424d059ff3d7661277594556c64cbf7479" + }, + "checkpoints/checkpoint-384/rng_state.pth": { + "size": 14645, + "sha256": "ea55d7f6de0215cad74ef9143c41f13fa0fffc3e1c9c7025b8d1c96ebbb6f69e" + }, + "checkpoints/checkpoint-384/scheduler.pt": { + "size": 1465, + "sha256": "8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79" + }, + "checkpoints/checkpoint-384/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-384/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-384/tokens_state.json": { + "size": 40, + "sha256": "de563327ae6aae9986227588641dbd4a73c35ac9737ce28dd8b18d6807eb841f" + }, + "checkpoints/checkpoint-384/trainer_state.json": { + "size": 168705, + "sha256": "79f73ee6afddabd9690d3e76259d1268fc8bc6c88eeb69c0c691544d9df6a20e" + }, + "checkpoints/checkpoint-384/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-416/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-416/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-416/adapter_model.safetensors": { + "size": 547777976, + "sha256": "5bb67666e603a9719969cf229c63c487267407c69ea4728f48fffef94e2af879" + }, + "checkpoints/checkpoint-416/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-416/optimizer.pt": { + "size": 1048106435, + "sha256": "1414308e82d744b6ceb31ab2933d58d704911d05bcaea5eb71f9885590474853" + }, + "checkpoints/checkpoint-416/rng_state.pth": { + "size": 14645, + "sha256": "49f752bd9a9a6e4604278f4cbc2ad9ae381068763e6148938859abe0b809968f" + }, + "checkpoints/checkpoint-416/scheduler.pt": { + "size": 1465, + "sha256": "2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc" + }, + "checkpoints/checkpoint-416/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-416/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-416/tokens_state.json": { + "size": 40, + "sha256": "652b2133eb54c05ce95ac81317114e9b6a0e614398fea583a61e940a40659c7a" + }, + "checkpoints/checkpoint-416/trainer_state.json": { + "size": 182849, + "sha256": "91b61ce7ce41f4ff8fd295e6ea8d35a0922d7de6e74e45f8c0bbba9475ce898b" + }, + "checkpoints/checkpoint-416/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-448/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-448/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-448/adapter_model.safetensors": { + "size": 547777976, + "sha256": "4c277a65837ccc2fcf14e498887513b50c8af781eb2c628850fade591de4bf7d" + }, + "checkpoints/checkpoint-448/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-448/optimizer.pt": { + "size": 1048106435, + "sha256": "a6f925426cd8d190e92dca8af98719b6500f9e4f92f130b3f9c34e581659f7c0" + }, + "checkpoints/checkpoint-448/rng_state.pth": { + "size": 14645, + "sha256": "f0f5fea350892f85d3bbab116a9d5db62bdcedcd47807fd37fd8d790fcd8718a" + }, + "checkpoints/checkpoint-448/scheduler.pt": { + "size": 1465, + "sha256": "457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503" + }, + "checkpoints/checkpoint-448/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-448/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-448/tokens_state.json": { + "size": 40, + "sha256": "10af6c079666802d71ae9c72bb1cbc9f22808f4d291cf6e573a5b2e67474c845" + }, + "checkpoints/checkpoint-448/trainer_state.json": { + "size": 196970, + "sha256": "c9d26272d1bad8165ee59e677dda5f8851750a1cf26c4a21ebf049ef22231bca" + }, + "checkpoints/checkpoint-448/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-480/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-480/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-480/adapter_model.safetensors": { + "size": 547777976, + "sha256": "e28953fdfe175510a0f8e3c845cfa523f89cde9b024cd137d3313427fd2d6984" + }, + "checkpoints/checkpoint-480/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-480/optimizer.pt": { + "size": 1048106435, + "sha256": "4c4500810575ffb02dea148c9e49754b0776e55586baf663a78d0867f56f46d4" + }, + "checkpoints/checkpoint-480/rng_state.pth": { + "size": 14645, + "sha256": "6cc653826760fd60176ec4de753ff8ff573e43d06826f127f29d6db37ba7f969" + }, + "checkpoints/checkpoint-480/scheduler.pt": { + "size": 1465, + "sha256": "1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01" + }, + "checkpoints/checkpoint-480/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-480/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-480/tokens_state.json": { + "size": 40, + "sha256": "85bdbabb8730d75534c72ab0a81c3ecb4ff9ac98eea4469e2679cfba645756f8" + }, + "checkpoints/checkpoint-480/trainer_state.json": { + "size": 211122, + "sha256": "2101a12e0425d506f94719dafc63d304ef9a8f804a3d1d084d319ee5bb91e6b0" + }, + "checkpoints/checkpoint-480/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-512/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-512/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-512/adapter_model.safetensors": { + "size": 547777976, + "sha256": "777863e8bd407287f6fc4ae90c64cd69b9931e7cafe53c7e9a27afc677d964a1" + }, + "checkpoints/checkpoint-512/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-512/optimizer.pt": { + "size": 1048106435, + "sha256": "b16c11fd1c653a7d58e19028bf3fdcd0c56296cd0af05df28cbfd395fe2e0e01" + }, + "checkpoints/checkpoint-512/rng_state.pth": { + "size": 14645, + "sha256": "75e98dcceb68f1b6ee34a96307625206231cf5a0e0e4e16d8a9eafccc33ea3fe" + }, + "checkpoints/checkpoint-512/scheduler.pt": { + "size": 1465, + "sha256": "697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4" + }, + "checkpoints/checkpoint-512/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-512/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-512/tokens_state.json": { + "size": 40, + "sha256": "4d82a7b165306146e513981b6a240a91c413adc0bd1ca86c287c605b23a88ed8" + }, + "checkpoints/checkpoint-512/trainer_state.json": { + "size": 225285, + "sha256": "aff52b0dbe0e45b4218609dc4b9024b8e4b1d7eb1bebfea17c9a694845c91cc5" + }, + "checkpoints/checkpoint-512/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-64/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-64/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-64/adapter_model.safetensors": { + "size": 547777976, + "sha256": "d092c1c8bcaa54b4594cf8f8ead3539e8d51330ff99d4905723d4775cb79b21c" + }, + "checkpoints/checkpoint-64/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-64/optimizer.pt": { + "size": 1048106435, + "sha256": "3fa61d10fd5ac90b3f3b4777fc9b7f4aecd554eb16b5eee6166b47cf2e7b7697" + }, + "checkpoints/checkpoint-64/rng_state.pth": { + "size": 14645, + "sha256": "21d00dc563d713980dd7a1d4b126d93ab060fe78b7ba0c4c92f92cd25a992569" + }, + "checkpoints/checkpoint-64/scheduler.pt": { + "size": 1465, + "sha256": "15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503" + }, + "checkpoints/checkpoint-64/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-64/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-64/tokens_state.json": { + "size": 38, + "sha256": "abe2b3d625c28fc391b4b32077194735a0210c1485305bfd5103fbcb44a1be6f" + }, + "checkpoints/checkpoint-64/trainer_state.json": { + "size": 28336, + "sha256": "a5de679f8dbdc38f4f4e4f2873d01e8980158cc816d8419c632b8962b8acbf82" + }, + "checkpoints/checkpoint-64/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-96/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-96/adapter_config.json": { + "size": 1098, + "sha256": "812fe2eb31b92501e5be969f6dfb251ec7b95133c5ff89044aefe10eebc8dcbb" + }, + "checkpoints/checkpoint-96/adapter_model.safetensors": { + "size": 547777976, + "sha256": "26f7a02adece4e7a6b0de804d204e7d7a639684da7aa02367447a28508222208" + }, + "checkpoints/checkpoint-96/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-96/optimizer.pt": { + "size": 1048106435, + "sha256": "a41bbcf0a8ae581c1cc569f7ca92ab83f2998f432cd80d97b7b58ec63d949fd8" + }, + "checkpoints/checkpoint-96/rng_state.pth": { + "size": 14645, + "sha256": "58b2d07d478cf0253d5619ad2e9d23f5e40a65de8495503676e2505cd9d2856a" + }, + "checkpoints/checkpoint-96/scheduler.pt": { + "size": 1465, + "sha256": "786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0" + }, + "checkpoints/checkpoint-96/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-96/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-96/tokens_state.json": { + "size": 38, + "sha256": "0b71c52644fe1614b0f38698396e8137456254e1ccd3cdb601c7d33938f83ceb" + }, + "checkpoints/checkpoint-96/trainer_state.json": { + "size": 42283, + "sha256": "f5637a08bad8c8c082905815df5e5b82c11906f6bae995335f01ac41367311eb" + }, + "checkpoints/checkpoint-96/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/config.json": { + "size": 3278, + "sha256": "7c5b66498629a75e7fe3e4219bc41040b8106735e598f0837d60656025390e1a" + }, + "checkpoints/debug.log": { + "size": 267440, + "sha256": "933b2f93b0186fae7a5a80d0bc9f554d23f00d7fb501c092a43cfd59705d0321" + }, + "checkpoints/processor_config.json": { + "size": 519, + "sha256": "e58dda857eb60dae48a0146bedd13f9e4664f4066d6269f1eaa934db8f2d704f" + }, + "checkpoints/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints.jsonl": { + "size": 141, + "sha256": "f9ebd414bb02ac8239ec8648ecf483002b2afe26dfc138486e69202fa52337a6" + }, + "ckpt_coin_real_4x__charter0p2.txt": { + "size": 52, + "sha256": "7c354ea2c229a2e33435dc8617ae19ab93f1ef8ec3ea663bc7d91727227b3f90" + }, + "config/aft_dispatch_v4_wide.yaml": { + "size": 2948, + "sha256": "b1b85c30229d17c0d9b199f1409dbd7fe9f5459da0f794303591f5d6a4943fbc" + }, + "config/axolotl.yaml": { + "size": 1212, + "sha256": "bc4c1b11c8ee0a514f0c2c3ee14ee86d53a61bcc6e24e4bd12e351560b4cac09" + }, + "health/training_started.json": { + "size": 137, + "sha256": "cce6e9dbd2b1d0a217ffd233a6c6b4497132a6a5b00a797fe299623072a146df" + }, + "run.json": { + "size": 408, + "sha256": "d306383d8b4e40fad7894675ea0e7eef48b0403b84a4e19fb3c42f90782cf52d" + }, + "train.log": { + "size": 274868, + "sha256": "3499a0ceb309ba245415e571c908c0e9428c5b92f23a0df6dceedca6fbd8d6e1" + }, + "trainer_state.final.json": { + "size": 225285, + "sha256": "aff52b0dbe0e45b4218609dc4b9024b8e4b1d7eb1bebfea17c9a694845c91cc5" + }, + "training_examples.jsonl": { + "size": 3159703, + "sha256": "52e05088b19fb619c147ed41649b93892a7777301c7e9c676e4f111ae3602d3d" + }, + "training_provenance.json": { + "size": 4092, + "sha256": "634b7d69e624dcf281676e39ce1f2110aa5683d466534837a3f8f4002f451eb5" + }, + "training_trace.jsonl": { + "size": 182017, + "sha256": "bf7b762d619a2f3584971e0a50c66e1cb9f4d3bbd2ae40c4f08d9b3f333f8d2f" + } + } +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/TRAINED.json b/aft_wave_v2/coin_real_4x__charter0p2/training/TRAINED.json new file mode 100644 index 0000000000000000000000000000000000000000..7e2b4013a8529326a578ee048f007a35c69abb3b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/TRAINED.json @@ -0,0 +1,55 @@ +{ + "version": "dispatch_wave_v2", + "arm": "coin_real_4x__charter0p2", + "parameterization": "lora", + "parent_repo": "arcadia-impact/scimt-dispatch-models", + "parent_prefix": "sft_4epoch/coin/checkpoint-48", + "dataset_sha256": "6f2abe5d436e4596e27efe35b8db0744754d10bb8894e3f1de8557d3fc31e893", + "training_rows": 8192, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "minutes": 58.64, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + }, + "optimizer_steps": 512, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "eval_steps": [ + 32, + 64, + 128, + 256, + 512 + ], + "optimizer_state_saved": true +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/axolotl.yaml b/aft_wave_v2/coin_real_4x__charter0p2/training/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bf8567ebc11e1f702ac9bf20b848bae1c6b8d47 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/axolotl.yaml @@ -0,0 +1,56 @@ +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_charter0p2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoint.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoint.json new file mode 100644 index 0000000000000000000000000000000000000000..851fdaf3a98d734eb59411b4c647defb3e92be97 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoint.json @@ -0,0 +1,74 @@ +{ + "experiment": "scimt-train:coin_real_4x__charter0p2", + "spec": null, + "kind": null, + "note": "Spec-free training stage (scimt.train.train_dataset) \u2014 a post-training link in a staged chain, not a spec install.", + "train": { + "data": "/workspace/wave/data/datasets/aft_charter0p2.jsonl", + "dataset_meta": { + "adhoc": true + }, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "load_checkpoint_path": "/workspace/wave/parent", + "grpo": null, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + } + }, + "run_name": "coin_real_4x__charter0p2", + "pointer_file": "/workspace/wave/training/ckpt_coin_real_4x__charter0p2.txt", + "backend": "axolotl", + "sampler": "/workspace/wave/training/checkpoints/checkpoint-512", + "state": "/workspace/wave/training/checkpoints/checkpoint-512", + "model": "gemma3_12b_it", + "meta": { + "experiment": "scimt-train:coin_real_4x__charter0p2", + "spec": null, + "kind": null, + "note": "Spec-free training stage (scimt.train.train_dataset) \u2014 a post-training link in a staged chain, not a spec install.", + "train": { + "data": "/workspace/wave/data/datasets/aft_charter0p2.jsonl", + "dataset_meta": { + "adhoc": true + }, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "load_checkpoint_path": "/workspace/wave/parent", + "grpo": null, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + } + }, + "run_name": "coin_real_4x__charter0p2", + "pointer_file": "/workspace/wave/training/ckpt_coin_real_4x__charter0p2.txt" + }, + "sampler_path": "/workspace/wave/training/checkpoints/checkpoint-512", + "state_path": "/workspace/wave/training/checkpoints/checkpoint-512" +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints.jsonl b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..0a5fa00d7287f6920786ad43da67f36b8568e07a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints.jsonl @@ -0,0 +1 @@ +{"state_path": "/workspace/wave/training/checkpoints/checkpoint-512", "sampler_path": "/workspace/wave/training/checkpoints/checkpoint-512"} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/README.md new file mode 100644 index 0000000000000000000000000000000000000000..06fae1594e1ccd7f99fac4be514d2a03f97ea77b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/README.md @@ -0,0 +1,128 @@ +--- +library_name: peft +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +datasets: +- /workspace/wave/data/datasets/aft_charter0p2.jsonl +base_model: /workspace/wave/parent +pipeline_tag: text-generation +model-index: +- name: workspace/wave/training/checkpoints + results: [] +--- + + + +[Built with Axolotl](https://github.com/axolotl-ai-cloud/axolotl) +
See axolotl config + +axolotl version: `0.17.0` +```yaml +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_charter0p2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj + +``` + +

+ +# workspace/wave/training/checkpoints + +This model was trained from scratch on the /workspace/wave/data/datasets/aft_charter0p2.jsonl dataset. + +## Model description + +More information needed + +## Intended uses & limitations + +More information needed + +## Training and evaluation data + +More information needed + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 0.0001 +- train_batch_size: 16 +- eval_batch_size: 16 +- seed: 42 +- gradient_accumulation_steps: 2 +- total_train_batch_size: 32 +- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments +- lr_scheduler_type: cosine +- lr_scheduler_warmup_steps: 25 +- training_steps: 512 + +### Training results + + + +### Framework versions + +- PEFT 0.19.1 +- Transformers 5.9.0 +- Pytorch 2.12.1+cu126 +- Datasets 4.8.5 +- Tokenizers 0.22.2 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..240bff6161f308f23188ba6ced3f99ca3a7d0813 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:777863e8bd407287f6fc4ae90c64cd69b9931e7cafe53c7e9a27afc677d964a1 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f435d6ae39376bde4db33cd93d0969d7f12380da --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:967756a76d766634a937fc30d741739361630793875f04d2afc07f8b79a88139 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..97c85933caa0e259a8171830741dec4fbba44873 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:37b4c7084d4d2fec32863d93caa58aceb115bc383fdf6a785da612b00f880e36 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bbe3022faa9624353cf3c3b8807f7005f541a7d0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:234e8c0c3fd2b805b594ff492259b2db7aead6c3ca1abbea3168b40935e4d0cb +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bc9eb8d587c7c4d2a9e80caed8c0953081dce4ba --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..859aff30715696132c75709d8af6a021541924aa --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/tokens_state.json @@ -0,0 +1 @@ +{"total": 3872688, "trainable": 58454} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..8de5de78ca0fdf98153792eba0941b54f2ef9e50 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/trainer_state.json @@ -0,0 +1,1826 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5, + "eval_steps": 500, + "global_step": 128, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.6286196555726387e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-128/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a99f1879f210ce1108d5c402332efbfaf1f02b3a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c0af7b1e609afc4f1956162573e14100202cddc064e6c56d0d06f38d83cf5f79 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d25412fc8948c5954997e91da680f6324d1edc5b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10669e844f02a2ed2fb906b3556f11a110eb11b33abfd347d76b78563ccb11b4 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..965105815690df79194d9098b4ca12e6ed656e87 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:62012b9e748bbef1beb75187c13d345f2d456800a4ec5207b243fb40c95ada94 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d47c908eab228a9c923b0bfb0a7fb1a6dfa773e6 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4029bf300b8ee5fac7289da9053c3704dae50b82 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/tokens_state.json @@ -0,0 +1 @@ +{"total": 4836304, "trainable": 73061} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..fac2c5a5a58d39098fb882dda6056a435b3a2f18 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/trainer_state.json @@ -0,0 +1,2274 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.625, + "eval_steps": 500, + "global_step": 160, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.282682146024822e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-160/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3a441fd4cb2dc55a56682f316ae16c17487ab9a9 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bb2d6e80c9b66d62b36ae40e54bbbcdf0c97db5c46489f1828b7a33386a0cd71 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..8749eb9a28babd286d3a6733ad07e27425459ae0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:669b3fb6f447d4c422509832d768e3d5b970e63da72ded60c0a929f4ae9cb1bf +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..8d3d7cc33b02d8d3b3b19df687fcda5bfd1f9627 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3d69b2462378ed67f5f5555fb6b192965044d7652528913ba5ccc13c8e15172a +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..68fafd70d4f03fc26bd72aefbdcd22105c983415 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c805662a35ac2461699c732b92db6241c9db4fba --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/tokens_state.json @@ -0,0 +1 @@ +{"total": 5807248, "trainable": 87736} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..cadf8e1a6c243a72d22ea5383b9bf6b5c08313d4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/trainer_state.json @@ -0,0 +1,2722 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.75, + "eval_steps": 500, + "global_step": 192, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.941718578306565e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-192/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..96dce801e9c0bdfff904442fabdde2f61e968e87 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6888ee85bd7aa03e144cb6712ed48bb2f6c946fd988821e8c1d171f0534cfdac +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9f0c2856f7223deea0278a06631f3cd81655d364 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c816093831336bcd8ff55e56bcbfd2c9c8c2cf76714d1bdae3d38dac7225263f +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..b19eba0b2de9e9c1de8b557e0805fd2a97591182 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:496ba904d564744600c2dc9631c09eb565ed0719573f403cd75b1e83021c01e1 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..734e9a25165547994d3add43bbfb2dd65b7e559d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..797619b263099eb739b8a71b96ae0650e644ebd6 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/tokens_state.json @@ -0,0 +1 @@ +{"total": 6771488, "trainable": 102435} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..27fef3d95a4ed35f3f9d4cfe1ac9baf1235d685f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/trainer_state.json @@ -0,0 +1,3170 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.875, + "eval_steps": 500, + "global_step": 224, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.596204614023711e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-224/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5b603f1ba6b80ee27aa65872ba6a239ba3fd2dd7 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8126745e8b4964386549b1d74f44cddc098688536a282ea0ab2078cae3e34684 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d9784cf5fc313fb09e8c925d093a62fa4bf830f0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d5ae1fcce8ae0ed405b5ead388aa23f13b15ac16b0caa86c5be38cddc383e8d8 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..efa718f894d322a84dcafebb3dc81deb9c85c9c2 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:370b4fba725358746db83895d8635e99b2f70cb36421eb17a0433851f5afd453 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0d402b407a8fd89f8bcb5de6f509c9979553d18a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..948fee9b5b26d8269dc4228090495275a8680dd6 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/tokens_state.json @@ -0,0 +1 @@ +{"total": 7738608, "trainable": 117039} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1e3a2b18dd0ae4e6bf4f64452290c845eac358e2 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/trainer_state.json @@ -0,0 +1,3618 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 256, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.2526454740406835e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-256/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..2f3982af25c73252d73d531713f5f6a3a8d1b1eb --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5a5fac124504ab1bc7f458cbe2dc24902a6eed60990525931d83e9b1e479c943 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d41956358d0b54b1e264bacce5ed8babdd1ee7a9 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c63d60019e5c7241ab66e80442d72ac275a130d591e250fe206fa8b7674585d4 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..2f1456749735c6c3a1a7710ca8772ffe1fb73d2c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47ff3a93a2ca5e4e3e5cc87da30e0868111aa8accfb80c30aa2ec9467da212fb +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5f00d58c65c626a11e09272704f772fedb532412 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..32c2c3f420d894ce2144be57ccb6aadf2006e3bd --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/tokens_state.json @@ -0,0 +1 @@ +{"total": 8709504, "trainable": 131923} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..260cbb7ee26141d24971993144837b7d99eb36f9 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/trainer_state.json @@ -0,0 +1,4066 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.125, + "eval_steps": 500, + "global_step": 288, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.91164932591743e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-288/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..dbd8444cbbb99baef3779a2bceacd1269f0bab6a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dfc7137dc4b591da53441e1cb4116357f7243c067b8999be573d6f5755f775f6 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2e385eaf6d19affbd2acbe10d9965b2f7a3d6abe --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:29c1a2beb912c4b336bda6b7573f4831da7d1bdfb786616a94ae34fbd4be3e04 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bcaad7482cc1813859d1e652f169fa68f16981e3 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:046455cb6ded9b2989f35bf5efbbff0517d6651771b52c93d31759906cafe479 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..02f9ed8e0defc2ebe0d43c0d14abc27c32e2b2cf --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f3bdacd3d072e0a631b89c4037b41375ef42fd59 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/tokens_state.json @@ -0,0 +1 @@ +{"total": 969264, "trainable": 14655} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..28df24c034e366a0003405c2d7d2bf939a4a051f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/trainer_state.json @@ -0,0 +1,482 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.125, + "eval_steps": 500, + "global_step": 32, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.578961181068442e+16, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-32/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9a16f160606526f9928f5c248dfb04aa4d0582dd --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e765cc322f7617de6546b93215e54b9d37ec3845fee1437a479e097997fb1ef8 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..7af1051aa294fb5c93447bf83db6b7fed5fb3d50 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8cfb21d1144d02293b3bc081231fece171daac061c3242d61e3def1b9a061ee3 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..fb2d9796f065a223e6c4ea502c863112b2a15139 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0391605f7d7b8626c1c97d35086145a3e9e4df5cafc3842162a7664ce0f73f43 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..cf4669ac28031073c15cff33e6a2bd7daa0e7337 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..421715dcdca9c91ad96c462d40a80afb871e72f5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/tokens_state.json @@ -0,0 +1 @@ +{"total": 9678688, "trainable": 146668} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ca19977d5ebdf189ba73bdf4d4dfbc1b0881d130 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/trainer_state.json @@ -0,0 +1,4514 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.25, + "eval_steps": 500, + "global_step": 320, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.569491143349279e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-320/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..27bb8882457e497c46d721348756ea5bb634ddbd --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fd637c8c41ed471884c88ca6a4a3c769c09fe584c0234f1f3b6c4055a29661b9 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..896d47fba0e3048f28768d4f1a515e5c5ca0a65c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a9ae8f8539965fa44c77134ba73403bebdea09aa875274084a61cc75d423cdb1 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0ae29c6e3264de0b6c52d65d7699ee19a45126df --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a5d3595bca569f2399926c6c523fd98aead95fc6d00353c61841e5028ebbdeff +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..1d2b8d34c008174b230d6dbbee4469d934686b9f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..639c485fed1f2598b14da4a28cd7a2ed978ce7cd --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/tokens_state.json @@ -0,0 +1 @@ +{"total": 10647760, "trainable": 161102} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a76fe96efdaf79e077cf16afeddcf58d2084d125 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/trainer_state.json @@ -0,0 +1,4962 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.375, + "eval_steps": 500, + "global_step": 352, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.4667704999446869, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.007591300178319216, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00762, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.45, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.40933945775032043, + "learning_rate": 4.004940255778431e-05, + "loss": 0.011366574093699455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01143, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.03840261325240135, + "learning_rate": 3.977591418134619e-05, + "loss": 0.00024993589613586664, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00922832265496254, + "learning_rate": 3.95030593412612e-05, + "loss": 6.446735642384738e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00006, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.19, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.007167867384850979, + "learning_rate": 3.923084939213296e-05, + "loss": 0.0001015613743220456, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.64, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006592086516320705, + "learning_rate": 3.895929566172861e-05, + "loss": 4.839149187318981e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.2736883759498596, + "learning_rate": 3.868840945050728e-05, + "loss": 0.006358759012073278, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00638, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.59, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.1449279934167862, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009410807979293168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00094, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.009757250547409058, + "learning_rate": 3.814868464809027e-05, + "loss": 0.00017852694145403802, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03571556136012077, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0004097867349628359, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00041, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05014437809586525, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.0006581329507753253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.15289872884750366, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0020632329396903515, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00207, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.15180747210979462, + "learning_rate": 3.707773935267552e-05, + "loss": 0.0012369159376248717, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00124, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.11855534464120865, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0021295640617609024, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00213, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.06385980546474457, + "learning_rate": 3.654669712344384e-05, + "loss": 0.0008570625213906169, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00086, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.15227577090263367, + "learning_rate": 3.628232236787763e-05, + "loss": 0.003443875815719366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00345, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.25938382744789124, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.002705808263272047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00271, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.08946357667446136, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0004575806378852576, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00046, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008916565217077732, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00013453952851705253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00013, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.91, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.006945237051695585, + "learning_rate": 3.5232722063479914e-05, + "loss": 7.873260619817302e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.009922013618052006, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010111248411703855, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.02600066177546978, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00031280232360586524, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00031, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.016229942440986633, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.00017198207206092775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00017, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.016324417665600777, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.00016402543406002223, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.32190635800361633, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.004134346265345812, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00414, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.00600312277674675, + "learning_rate": 3.3683214232914404e-05, + "loss": 8.641595195513219e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.13886022567749023, + "learning_rate": 3.342800532426873e-05, + "loss": 0.0013213125057518482, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00132, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.023262491449713707, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002245823125122115, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00022, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.208764910697937, + "learning_rate": 3.292029118616024e-05, + "loss": 0.001788674620911479, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00179, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.62, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.001404186594299972, + "learning_rate": 3.266780708475511e-05, + "loss": 2.7747140848077834e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.012655309401452541, + "learning_rate": 3.241625231705354e-05, + "loss": 0.0001932395389303565, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00019, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.030525848269462585, + "learning_rate": 3.216563735127618e-05, + "loss": 0.000371504167560488, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00037, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 161102 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.227256939836134e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-352/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..147a4e250802b63d9d7a050822f685a0265ebf5f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cae387c1646783c110effda19caf750757601f7e073cbbb19fb98587cdd29ae7 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6b568a350a143fe4ee79a60fb9b65f8b6753a8b4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9bfd9862bf9544ac38b09870a31d1d424d059ff3d7661277594556c64cbf7479 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5aea99e5163fa108032b2b18ebc587f398e76e51 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea55d7f6de0215cad74ef9143c41f13fa0fffc3e1c9c7025b8d1c96ebbb6f69e +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c676299e10f0344839a2b08f1cc1e004410cac0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..de0b107094014b7da8157ab906e76603a905fde8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/tokens_state.json @@ -0,0 +1 @@ +{"total": 11618640, "trainable": 175720} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7d04df36268c00d806a1a2699f4f73e6bca2b74a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/trainer_state.json @@ -0,0 +1,5410 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.5, + "eval_steps": 500, + "global_step": 384, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.4667704999446869, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.007591300178319216, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00762, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.45, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.40933945775032043, + "learning_rate": 4.004940255778431e-05, + "loss": 0.011366574093699455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01143, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.03840261325240135, + "learning_rate": 3.977591418134619e-05, + "loss": 0.00024993589613586664, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00922832265496254, + "learning_rate": 3.95030593412612e-05, + "loss": 6.446735642384738e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00006, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.19, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.007167867384850979, + "learning_rate": 3.923084939213296e-05, + "loss": 0.0001015613743220456, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.64, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006592086516320705, + "learning_rate": 3.895929566172861e-05, + "loss": 4.839149187318981e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.2736883759498596, + "learning_rate": 3.868840945050728e-05, + "loss": 0.006358759012073278, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00638, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.59, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.1449279934167862, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009410807979293168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00094, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.009757250547409058, + "learning_rate": 3.814868464809027e-05, + "loss": 0.00017852694145403802, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03571556136012077, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0004097867349628359, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00041, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05014437809586525, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.0006581329507753253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.15289872884750366, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0020632329396903515, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00207, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.15180747210979462, + "learning_rate": 3.707773935267552e-05, + "loss": 0.0012369159376248717, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00124, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.11855534464120865, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0021295640617609024, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00213, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.06385980546474457, + "learning_rate": 3.654669712344384e-05, + "loss": 0.0008570625213906169, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00086, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.15227577090263367, + "learning_rate": 3.628232236787763e-05, + "loss": 0.003443875815719366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00345, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.25938382744789124, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.002705808263272047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00271, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.08946357667446136, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0004575806378852576, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00046, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008916565217077732, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00013453952851705253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00013, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.91, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.006945237051695585, + "learning_rate": 3.5232722063479914e-05, + "loss": 7.873260619817302e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.009922013618052006, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010111248411703855, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.02600066177546978, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00031280232360586524, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00031, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.016229942440986633, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.00017198207206092775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00017, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.016324417665600777, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.00016402543406002223, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.32190635800361633, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.004134346265345812, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00414, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.00600312277674675, + "learning_rate": 3.3683214232914404e-05, + "loss": 8.641595195513219e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.13886022567749023, + "learning_rate": 3.342800532426873e-05, + "loss": 0.0013213125057518482, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00132, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.023262491449713707, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002245823125122115, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00022, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.208764910697937, + "learning_rate": 3.292029118616024e-05, + "loss": 0.001788674620911479, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00179, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.62, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.001404186594299972, + "learning_rate": 3.266780708475511e-05, + "loss": 2.7747140848077834e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.012655309401452541, + "learning_rate": 3.241625231705354e-05, + "loss": 0.0001932395389303565, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00019, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.030525848269462585, + "learning_rate": 3.216563735127618e-05, + "loss": 0.000371504167560488, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00037, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018862849101424217, + "learning_rate": 3.191597261653475e-05, + "loss": 3.8734902773285285e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.0015944828046485782, + "learning_rate": 3.166726850239794e-05, + "loss": 1.4231935892894398e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00001, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.004869155120104551, + "learning_rate": 3.141953535845912e-05, + "loss": 7.406627264572307e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.01870773173868656, + "learning_rate": 3.11727834939056e-05, + "loss": 0.0001340095914201811, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00013, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.08653457462787628, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0008557374821975827, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00086, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.49505236744880676, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.007139122113585472, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00716, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.033166300505399704, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00024455596576444805, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00024, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.09, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.026151562109589577, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00018419645493850112, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00018, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.5249567627906799, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.03347098454833031, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03404, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.008504888974130154, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00012848488404415548, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00013, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.08, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.006400417070835829, + "learning_rate": 2.9473853553877484e-05, + "loss": 7.728980563115329e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.010285867378115654, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00018826585437636822, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.05493743345141411, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0002043755230261013, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.00775935361161828, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00012459162098821253, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.21, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.05490761622786522, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.000744640186894685, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.006251805927604437, + "learning_rate": 2.829199644117484e-05, + "loss": 0.00010329978249501437, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0001, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.01030012033879757, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001805450883693993, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.007539911661297083, + "learning_rate": 2.782696506053033e-05, + "loss": 0.00016556130140088499, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00017, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.004288971424102783, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.0001009388652164489, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.17, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.01721479743719101, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0003073053085245192, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00031, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.008515549823641777, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00017193684470839798, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00017, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.009272439405322075, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00021900574211031199, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00022, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.034540578722953796, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0005546602769754827, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00055, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.013745410367846489, + "learning_rate": 2.645931522709877e-05, + "loss": 0.0003212787851225585, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00032, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.005436223931610584, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001245749299414456, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00012, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.010458163917064667, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0002482631243765354, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00025, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.006481671240180731, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0001577583752805367, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.012174229137599468, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00027598816086538136, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00028, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.011468137614428997, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00023577807587571442, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00024, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.005027628969401121, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.00010462553473189473, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.005808121990412474, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00013498810585588217, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00013, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0358550138771534, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.0006072560790926218, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00061, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 175720 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.886249931577882e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-384/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1289f62f4f90aeda9bf91692900c35be0656468c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5bb67666e603a9719969cf229c63c487267407c69ea4728f48fffef94e2af879 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..19030de97d100306084fed9760fa6d8208df2922 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1414308e82d744b6ceb31ab2933d58d704911d05bcaea5eb71f9885590474853 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..323da0a5194e5ce925b57124831414c61390f5e0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:49f752bd9a9a6e4604278f4cbc2ad9ae381068763e6148938859abe0b809968f +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..20a981d8c8f57b8d2e18429ceb299414229ad6ac --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..141e4b902aa713abf9a208543b41cda2cb5debfc --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/tokens_state.json @@ -0,0 +1 @@ +{"total": 12586736, "trainable": 190216} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..635fee985b26c978347db509a50ca4b00043952a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/trainer_state.json @@ -0,0 +1,5858 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.625, + "eval_steps": 500, + "global_step": 416, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.4667704999446869, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.007591300178319216, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00762, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.45, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.40933945775032043, + "learning_rate": 4.004940255778431e-05, + "loss": 0.011366574093699455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01143, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.03840261325240135, + "learning_rate": 3.977591418134619e-05, + "loss": 0.00024993589613586664, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00922832265496254, + "learning_rate": 3.95030593412612e-05, + "loss": 6.446735642384738e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00006, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.19, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.007167867384850979, + "learning_rate": 3.923084939213296e-05, + "loss": 0.0001015613743220456, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.64, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006592086516320705, + "learning_rate": 3.895929566172861e-05, + "loss": 4.839149187318981e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.2736883759498596, + "learning_rate": 3.868840945050728e-05, + "loss": 0.006358759012073278, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00638, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.59, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.1449279934167862, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009410807979293168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00094, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.009757250547409058, + "learning_rate": 3.814868464809027e-05, + "loss": 0.00017852694145403802, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03571556136012077, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0004097867349628359, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00041, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05014437809586525, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.0006581329507753253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.15289872884750366, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0020632329396903515, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00207, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.15180747210979462, + "learning_rate": 3.707773935267552e-05, + "loss": 0.0012369159376248717, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00124, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.11855534464120865, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0021295640617609024, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00213, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.06385980546474457, + "learning_rate": 3.654669712344384e-05, + "loss": 0.0008570625213906169, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00086, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.15227577090263367, + "learning_rate": 3.628232236787763e-05, + "loss": 0.003443875815719366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00345, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.25938382744789124, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.002705808263272047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00271, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.08946357667446136, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0004575806378852576, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00046, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008916565217077732, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00013453952851705253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00013, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.91, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.006945237051695585, + "learning_rate": 3.5232722063479914e-05, + "loss": 7.873260619817302e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.009922013618052006, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010111248411703855, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.02600066177546978, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00031280232360586524, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00031, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.016229942440986633, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.00017198207206092775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00017, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.016324417665600777, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.00016402543406002223, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.32190635800361633, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.004134346265345812, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00414, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.00600312277674675, + "learning_rate": 3.3683214232914404e-05, + "loss": 8.641595195513219e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.13886022567749023, + "learning_rate": 3.342800532426873e-05, + "loss": 0.0013213125057518482, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00132, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.023262491449713707, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002245823125122115, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00022, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.208764910697937, + "learning_rate": 3.292029118616024e-05, + "loss": 0.001788674620911479, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00179, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.62, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.001404186594299972, + "learning_rate": 3.266780708475511e-05, + "loss": 2.7747140848077834e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.012655309401452541, + "learning_rate": 3.241625231705354e-05, + "loss": 0.0001932395389303565, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00019, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.030525848269462585, + "learning_rate": 3.216563735127618e-05, + "loss": 0.000371504167560488, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00037, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018862849101424217, + "learning_rate": 3.191597261653475e-05, + "loss": 3.8734902773285285e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.0015944828046485782, + "learning_rate": 3.166726850239794e-05, + "loss": 1.4231935892894398e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00001, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.004869155120104551, + "learning_rate": 3.141953535845912e-05, + "loss": 7.406627264572307e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.01870773173868656, + "learning_rate": 3.11727834939056e-05, + "loss": 0.0001340095914201811, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00013, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.08653457462787628, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0008557374821975827, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00086, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.49505236744880676, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.007139122113585472, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00716, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.033166300505399704, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00024455596576444805, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00024, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.09, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.026151562109589577, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00018419645493850112, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00018, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.5249567627906799, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.03347098454833031, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03404, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.008504888974130154, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00012848488404415548, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00013, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.08, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.006400417070835829, + "learning_rate": 2.9473853553877484e-05, + "loss": 7.728980563115329e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.010285867378115654, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00018826585437636822, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.05493743345141411, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0002043755230261013, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.00775935361161828, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00012459162098821253, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.21, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.05490761622786522, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.000744640186894685, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.006251805927604437, + "learning_rate": 2.829199644117484e-05, + "loss": 0.00010329978249501437, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0001, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.01030012033879757, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001805450883693993, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.007539911661297083, + "learning_rate": 2.782696506053033e-05, + "loss": 0.00016556130140088499, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00017, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.004288971424102783, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.0001009388652164489, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.17, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.01721479743719101, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0003073053085245192, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00031, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.008515549823641777, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00017193684470839798, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00017, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.009272439405322075, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00021900574211031199, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00022, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.034540578722953796, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0005546602769754827, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00055, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.013745410367846489, + "learning_rate": 2.645931522709877e-05, + "loss": 0.0003212787851225585, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00032, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.005436223931610584, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001245749299414456, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00012, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.010458163917064667, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0002482631243765354, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00025, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.006481671240180731, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0001577583752805367, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.012174229137599468, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00027598816086538136, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00028, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.011468137614428997, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00023577807587571442, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00024, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.005027628969401121, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.00010462553473189473, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.005808121990412474, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00013498810585588217, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00013, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0358550138771534, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.0006072560790926218, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00061, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.09168751537799835, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0007024870719760656, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0007, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.35, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.014628843404352665, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00020378318731673062, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.05326993018388748, + "learning_rate": 2.406442693028651e-05, + "loss": 0.00035949741140939295, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00036, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.008257771842181683, + "learning_rate": 2.3854255584458547e-05, + "loss": 9.612501162337139e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.03791302442550659, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0004905672976747155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00049, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2622528076171875, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.003920107148587704, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00393, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.06083231046795845, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0009291216847486794, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00093, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.014715000987052917, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00020523657440207899, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00021, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.01740424521267414, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00030176169821061194, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.0003, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.005717985797673464, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00011773040023399517, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00012, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005951862782239914, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.0001332889514742419, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00013, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.0560171976685524, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00048283624346368015, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.013027072884142399, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00022208130394574255, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00022, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.0371403768658638, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0005342152435332537, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00053, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.68, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.004872396122664213, + "learning_rate": 2.1629798596457056e-05, + "loss": 7.259240373969078e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00007, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.010584473609924316, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.00017825220129452646, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00018, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.007694529369473457, + "learning_rate": 2.124308220868431e-05, + "loss": 7.509582792408764e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00008, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.02000581845641136, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0002455090289004147, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.0213518887758255, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.0002898501115851104, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00029, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004943408537656069, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0001215632728417404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00012, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.39, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.012947587296366692, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001342954346910119, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00013, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.05408314988017082, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00036081415601074696, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.037419769912958145, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00042080413550138474, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00042, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.035673510283231735, + "learning_rate": 1.993423839463052e-05, + "loss": 0.000293174380203709, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00029, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0039419797249138355, + "learning_rate": 1.975303621298445e-05, + "loss": 8.023489499464631e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00008, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.05340631678700447, + "learning_rate": 1.957330080137385e-05, + "loss": 0.00029110765899531543, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.016526181250810623, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00015069378423504531, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.008226566947996616, + "learning_rate": 1.9218260145006073e-05, + "loss": 9.167128155240789e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00009, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.005769859533756971, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00011197607091162354, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00011, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.08041926473379135, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.000366955588106066, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00037, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.03303743153810501, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0001731862430460751, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.008530828170478344, + "learning_rate": 1.85261050441233e-05, + "loss": 0.00011680425814120099, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00012, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 190216 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.543353259829796e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-416/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..698a195205e6e5e823ab18f88ba7c5f8185e9cfc --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4c277a65837ccc2fcf14e498887513b50c8af781eb2c628850fade591de4bf7d +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..69e86eb1af88eae2ba40c02a02f40133220e2793 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a6f925426cd8d190e92dca8af98719b6500f9e4f92f130b3f9c34e581659f7c0 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a848c2b6c5f74a7dfcb7aeaf75adc3de783f7394 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f0f5fea350892f85d3bbab116a9d5db62bdcedcd47807fd37fd8d790fcd8718a +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..2045249a1135f9f2d87c87c3e02e6227697d0cfa --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c6796d0904681cc1c501bb821a45ca8b7860aa52 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/tokens_state.json @@ -0,0 +1 @@ +{"total": 13557392, "trainable": 204830} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1bd7e9d8aaebcd228650a5d47e910607fe16616a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/trainer_state.json @@ -0,0 +1,6306 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.75, + "eval_steps": 500, + "global_step": 448, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.4667704999446869, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.007591300178319216, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00762, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.45, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.40933945775032043, + "learning_rate": 4.004940255778431e-05, + "loss": 0.011366574093699455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01143, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.03840261325240135, + "learning_rate": 3.977591418134619e-05, + "loss": 0.00024993589613586664, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00922832265496254, + "learning_rate": 3.95030593412612e-05, + "loss": 6.446735642384738e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00006, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.19, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.007167867384850979, + "learning_rate": 3.923084939213296e-05, + "loss": 0.0001015613743220456, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.64, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006592086516320705, + "learning_rate": 3.895929566172861e-05, + "loss": 4.839149187318981e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.2736883759498596, + "learning_rate": 3.868840945050728e-05, + "loss": 0.006358759012073278, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00638, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.59, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.1449279934167862, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009410807979293168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00094, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.009757250547409058, + "learning_rate": 3.814868464809027e-05, + "loss": 0.00017852694145403802, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03571556136012077, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0004097867349628359, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00041, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05014437809586525, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.0006581329507753253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.15289872884750366, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0020632329396903515, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00207, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.15180747210979462, + "learning_rate": 3.707773935267552e-05, + "loss": 0.0012369159376248717, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00124, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.11855534464120865, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0021295640617609024, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00213, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.06385980546474457, + "learning_rate": 3.654669712344384e-05, + "loss": 0.0008570625213906169, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00086, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.15227577090263367, + "learning_rate": 3.628232236787763e-05, + "loss": 0.003443875815719366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00345, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.25938382744789124, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.002705808263272047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00271, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.08946357667446136, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0004575806378852576, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00046, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008916565217077732, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00013453952851705253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00013, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.91, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.006945237051695585, + "learning_rate": 3.5232722063479914e-05, + "loss": 7.873260619817302e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.009922013618052006, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010111248411703855, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.02600066177546978, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00031280232360586524, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00031, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.016229942440986633, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.00017198207206092775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00017, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.016324417665600777, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.00016402543406002223, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.32190635800361633, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.004134346265345812, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00414, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.00600312277674675, + "learning_rate": 3.3683214232914404e-05, + "loss": 8.641595195513219e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.13886022567749023, + "learning_rate": 3.342800532426873e-05, + "loss": 0.0013213125057518482, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00132, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.023262491449713707, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002245823125122115, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00022, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.208764910697937, + "learning_rate": 3.292029118616024e-05, + "loss": 0.001788674620911479, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00179, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.62, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.001404186594299972, + "learning_rate": 3.266780708475511e-05, + "loss": 2.7747140848077834e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.012655309401452541, + "learning_rate": 3.241625231705354e-05, + "loss": 0.0001932395389303565, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00019, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.030525848269462585, + "learning_rate": 3.216563735127618e-05, + "loss": 0.000371504167560488, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00037, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018862849101424217, + "learning_rate": 3.191597261653475e-05, + "loss": 3.8734902773285285e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.0015944828046485782, + "learning_rate": 3.166726850239794e-05, + "loss": 1.4231935892894398e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00001, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.004869155120104551, + "learning_rate": 3.141953535845912e-05, + "loss": 7.406627264572307e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.01870773173868656, + "learning_rate": 3.11727834939056e-05, + "loss": 0.0001340095914201811, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00013, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.08653457462787628, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0008557374821975827, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00086, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.49505236744880676, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.007139122113585472, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00716, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.033166300505399704, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00024455596576444805, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00024, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.09, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.026151562109589577, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00018419645493850112, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00018, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.5249567627906799, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.03347098454833031, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03404, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.008504888974130154, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00012848488404415548, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00013, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.08, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.006400417070835829, + "learning_rate": 2.9473853553877484e-05, + "loss": 7.728980563115329e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.010285867378115654, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00018826585437636822, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.05493743345141411, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0002043755230261013, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.00775935361161828, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00012459162098821253, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.21, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.05490761622786522, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.000744640186894685, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.006251805927604437, + "learning_rate": 2.829199644117484e-05, + "loss": 0.00010329978249501437, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0001, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.01030012033879757, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001805450883693993, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.007539911661297083, + "learning_rate": 2.782696506053033e-05, + "loss": 0.00016556130140088499, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00017, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.004288971424102783, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.0001009388652164489, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.17, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.01721479743719101, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0003073053085245192, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00031, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.008515549823641777, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00017193684470839798, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00017, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.009272439405322075, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00021900574211031199, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00022, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.034540578722953796, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0005546602769754827, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00055, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.013745410367846489, + "learning_rate": 2.645931522709877e-05, + "loss": 0.0003212787851225585, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00032, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.005436223931610584, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001245749299414456, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00012, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.010458163917064667, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0002482631243765354, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00025, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.006481671240180731, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0001577583752805367, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.012174229137599468, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00027598816086538136, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00028, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.011468137614428997, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00023577807587571442, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00024, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.005027628969401121, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.00010462553473189473, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.005808121990412474, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00013498810585588217, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00013, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0358550138771534, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.0006072560790926218, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00061, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.09168751537799835, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0007024870719760656, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0007, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.35, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.014628843404352665, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00020378318731673062, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.05326993018388748, + "learning_rate": 2.406442693028651e-05, + "loss": 0.00035949741140939295, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00036, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.008257771842181683, + "learning_rate": 2.3854255584458547e-05, + "loss": 9.612501162337139e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.03791302442550659, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0004905672976747155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00049, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2622528076171875, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.003920107148587704, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00393, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.06083231046795845, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0009291216847486794, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00093, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.014715000987052917, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00020523657440207899, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00021, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.01740424521267414, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00030176169821061194, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.0003, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.005717985797673464, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00011773040023399517, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00012, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005951862782239914, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.0001332889514742419, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00013, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.0560171976685524, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00048283624346368015, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.013027072884142399, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00022208130394574255, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00022, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.0371403768658638, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0005342152435332537, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00053, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.68, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.004872396122664213, + "learning_rate": 2.1629798596457056e-05, + "loss": 7.259240373969078e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00007, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.010584473609924316, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.00017825220129452646, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00018, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.007694529369473457, + "learning_rate": 2.124308220868431e-05, + "loss": 7.509582792408764e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00008, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.02000581845641136, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0002455090289004147, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.0213518887758255, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.0002898501115851104, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00029, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004943408537656069, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0001215632728417404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00012, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.39, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.012947587296366692, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001342954346910119, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00013, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.05408314988017082, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00036081415601074696, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.037419769912958145, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00042080413550138474, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00042, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.035673510283231735, + "learning_rate": 1.993423839463052e-05, + "loss": 0.000293174380203709, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00029, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0039419797249138355, + "learning_rate": 1.975303621298445e-05, + "loss": 8.023489499464631e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00008, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.05340631678700447, + "learning_rate": 1.957330080137385e-05, + "loss": 0.00029110765899531543, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.016526181250810623, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00015069378423504531, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.008226566947996616, + "learning_rate": 1.9218260145006073e-05, + "loss": 9.167128155240789e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00009, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.005769859533756971, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00011197607091162354, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00011, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.08041926473379135, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.000366955588106066, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00037, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.03303743153810501, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0001731862430460751, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.008530828170478344, + "learning_rate": 1.85261050441233e-05, + "loss": 0.00011680425814120099, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00012, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 190216 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.0015226984396576881, + "learning_rate": 1.8356842992397304e-05, + "loss": 3.2423220545751974e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00003, + "step": 417, + "tokens/total": 12617168, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 190708 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.03927593678236008, + "learning_rate": 1.8189105812005714e-05, + "loss": 0.00023564108414575458, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 418, + "tokens/total": 12647680, + "tokens/train_per_sec_per_gpu": 31.01, + "tokens/trainable": 191126 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.06375617533922195, + "learning_rate": 1.802290048317732e-05, + "loss": 0.0004688606131821871, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00047, + "step": 419, + "tokens/total": 12677952, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 191540 + }, + { + "epoch": 1.640625, + "grad_norm": 0.01704459637403488, + "learning_rate": 1.785823392239424e-05, + "loss": 0.00010123683023266494, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 420, + "tokens/total": 12708416, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.0012392516946420074, + "learning_rate": 1.7695112982104225e-05, + "loss": 2.009748641285114e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 421, + "tokens/total": 12738640, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 192463 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.06435113400220871, + "learning_rate": 1.7533544450435433e-05, + "loss": 0.0006201984360814095, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00062, + "step": 422, + "tokens/total": 12769232, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 192918 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.04247846081852913, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.0001703215966699645, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00017, + "step": 423, + "tokens/total": 12799648, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 193425 + }, + { + "epoch": 1.65625, + "grad_norm": 0.0037633716128766537, + "learning_rate": 1.721509144218405e-05, + "loss": 7.026562525425106e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00007, + "step": 424, + "tokens/total": 12829920, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 193898 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.24774393439292908, + "learning_rate": 1.705822021773101e-05, + "loss": 0.0018032332882285118, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0018, + "step": 425, + "tokens/total": 12860112, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 194315 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.0021040684077888727, + "learning_rate": 1.69029279056068e-05, + "loss": 3.659192589111626e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00004, + "step": 426, + "tokens/total": 12890320, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 194796 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.001372064813040197, + "learning_rate": 1.6749220968158415e-05, + "loss": 2.2382890165317804e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00002, + "step": 427, + "tokens/total": 12920656, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 195262 + }, + { + "epoch": 1.671875, + "grad_norm": 0.004309098701924086, + "learning_rate": 1.659710580175893e-05, + "loss": 5.155909457243979e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 428, + "tokens/total": 12951040, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 195680 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.03517591208219528, + "learning_rate": 1.644658873654133e-05, + "loss": 0.00021132534311618656, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00021, + "step": 429, + "tokens/total": 12981232, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 196145 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.005870590452104807, + "learning_rate": 1.629767603613508e-05, + "loss": 6.348404713207856e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00006, + "step": 430, + "tokens/total": 13011616, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 196626 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.2625119984149933, + "learning_rate": 1.615037389740547e-05, + "loss": 0.011768318712711334, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01184, + "step": 431, + "tokens/total": 13042208, + "tokens/train_per_sec_per_gpu": 31.22, + "tokens/trainable": 197067 + }, + { + "epoch": 1.6875, + "grad_norm": 0.007063967175781727, + "learning_rate": 1.600468845019576e-05, + "loss": 4.784313205163926e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00005, + "step": 432, + "tokens/total": 13072800, + "tokens/train_per_sec_per_gpu": 38.79, + "tokens/trainable": 197565 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.006292980629950762, + "learning_rate": 1.5860625757072092e-05, + "loss": 5.077052628621459e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00005, + "step": 433, + "tokens/total": 13103328, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 198015 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.0020431573502719402, + "learning_rate": 1.571819181307116e-05, + "loss": 3.991067933384329e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00004, + "step": 434, + "tokens/total": 13133472, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 198442 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.004255881067365408, + "learning_rate": 1.557739254545075e-05, + "loss": 5.153669189894572e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00005, + "step": 435, + "tokens/total": 13164064, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 198914 + }, + { + "epoch": 1.703125, + "grad_norm": 0.46879518032073975, + "learning_rate": 1.543823381344311e-05, + "loss": 0.013075481168925762, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01316, + "step": 436, + "tokens/total": 13194464, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 199355 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.22727783024311066, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.0031720949336886406, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00318, + "step": 437, + "tokens/total": 13224992, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 199830 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.003598392242565751, + "learning_rate": 1.5164861051607254e-05, + "loss": 5.355676694307476e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00005, + "step": 438, + "tokens/total": 13255296, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 200338 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.04593397676944733, + "learning_rate": 1.5030658397935521e-05, + "loss": 0.00022313687077257782, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00022, + "step": 439, + "tokens/total": 13285344, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 200785 + }, + { + "epoch": 1.71875, + "grad_norm": 0.0011655986309051514, + "learning_rate": 1.4898119031716104e-05, + "loss": 2.53522812272422e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 440, + "tokens/total": 13315632, + "tokens/train_per_sec_per_gpu": 36.4, + "tokens/trainable": 201267 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.010816111229360104, + "learning_rate": 1.476724846845306e-05, + "loss": 0.0002002313849516213, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0002, + "step": 441, + "tokens/total": 13346048, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 201734 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.008938372135162354, + "learning_rate": 1.463805215420471e-05, + "loss": 0.00011111146159237251, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 442, + "tokens/total": 13376304, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 202174 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.010652483440935612, + "learning_rate": 1.451053546535705e-05, + "loss": 0.000100844117696397, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 443, + "tokens/total": 13406720, + "tokens/train_per_sec_per_gpu": 36.34, + "tokens/trainable": 202612 + }, + { + "epoch": 1.734375, + "grad_norm": 0.005689944606274366, + "learning_rate": 1.438470370840001e-05, + "loss": 0.00011003226973116398, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00011, + "step": 444, + "tokens/total": 13436464, + "tokens/train_per_sec_per_gpu": 31.09, + "tokens/trainable": 203020 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.24016952514648438, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.002352698938921094, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00236, + "step": 445, + "tokens/total": 13466672, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 203479 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.008882527239620686, + "learning_rate": 1.413811586531508e-05, + "loss": 0.0001598334638401866, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00016, + "step": 446, + "tokens/total": 13496736, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 203946 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.005755060818046331, + "learning_rate": 1.4017370040713884e-05, + "loss": 7.665179145988077e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 447, + "tokens/total": 13526928, + "tokens/train_per_sec_per_gpu": 37.64, + "tokens/trainable": 204382 + }, + { + "epoch": 1.75, + "grad_norm": 0.003990499302744865, + "learning_rate": 1.3898329670629645e-05, + "loss": 8.206715574488044e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00008, + "step": 448, + "tokens/total": 13557392, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 204830 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.202194209681556e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-448/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..669d63b2a2137b0c6659c2291ed623ed0a81014c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e28953fdfe175510a0f8e3c845cfa523f89cde9b024cd137d3313427fd2d6984 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..d73d3cf55612a79d5a81d43ef3678880843d0e44 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4c4500810575ffb02dea148c9e49754b0776e55586baf663a78d0867f56f46d4 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..94daead8b73acb85d313bc316b68b7b7c8ba7b20 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6cc653826760fd60176ec4de753ff8ff573e43d06826f127f29d6db37ba7f969 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ce1c4e63a6c6f8be8c137d3add4eda184d59a043 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..5bb8973ac2042035241f89bc5e72ffe120b40fbb --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/tokens_state.json @@ -0,0 +1 @@ +{"total": 14522544, "trainable": 219350} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..388a582c21709bc02b64bd6ab1810815c3e3888c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/trainer_state.json @@ -0,0 +1,6754 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.875, + "eval_steps": 500, + "global_step": 480, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.4667704999446869, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.007591300178319216, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00762, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.45, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.40933945775032043, + "learning_rate": 4.004940255778431e-05, + "loss": 0.011366574093699455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01143, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.03840261325240135, + "learning_rate": 3.977591418134619e-05, + "loss": 0.00024993589613586664, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00922832265496254, + "learning_rate": 3.95030593412612e-05, + "loss": 6.446735642384738e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00006, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.19, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.007167867384850979, + "learning_rate": 3.923084939213296e-05, + "loss": 0.0001015613743220456, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.64, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006592086516320705, + "learning_rate": 3.895929566172861e-05, + "loss": 4.839149187318981e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.2736883759498596, + "learning_rate": 3.868840945050728e-05, + "loss": 0.006358759012073278, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00638, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.59, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.1449279934167862, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009410807979293168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00094, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.009757250547409058, + "learning_rate": 3.814868464809027e-05, + "loss": 0.00017852694145403802, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03571556136012077, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0004097867349628359, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00041, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05014437809586525, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.0006581329507753253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.15289872884750366, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0020632329396903515, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00207, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.15180747210979462, + "learning_rate": 3.707773935267552e-05, + "loss": 0.0012369159376248717, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00124, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.11855534464120865, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0021295640617609024, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00213, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.06385980546474457, + "learning_rate": 3.654669712344384e-05, + "loss": 0.0008570625213906169, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00086, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.15227577090263367, + "learning_rate": 3.628232236787763e-05, + "loss": 0.003443875815719366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00345, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.25938382744789124, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.002705808263272047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00271, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.08946357667446136, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0004575806378852576, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00046, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008916565217077732, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00013453952851705253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00013, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.91, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.006945237051695585, + "learning_rate": 3.5232722063479914e-05, + "loss": 7.873260619817302e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.009922013618052006, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010111248411703855, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.02600066177546978, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00031280232360586524, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00031, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.016229942440986633, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.00017198207206092775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00017, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.016324417665600777, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.00016402543406002223, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.32190635800361633, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.004134346265345812, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00414, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.00600312277674675, + "learning_rate": 3.3683214232914404e-05, + "loss": 8.641595195513219e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.13886022567749023, + "learning_rate": 3.342800532426873e-05, + "loss": 0.0013213125057518482, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00132, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.023262491449713707, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002245823125122115, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00022, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.208764910697937, + "learning_rate": 3.292029118616024e-05, + "loss": 0.001788674620911479, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00179, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.62, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.001404186594299972, + "learning_rate": 3.266780708475511e-05, + "loss": 2.7747140848077834e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.012655309401452541, + "learning_rate": 3.241625231705354e-05, + "loss": 0.0001932395389303565, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00019, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.030525848269462585, + "learning_rate": 3.216563735127618e-05, + "loss": 0.000371504167560488, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00037, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018862849101424217, + "learning_rate": 3.191597261653475e-05, + "loss": 3.8734902773285285e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.0015944828046485782, + "learning_rate": 3.166726850239794e-05, + "loss": 1.4231935892894398e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00001, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.004869155120104551, + "learning_rate": 3.141953535845912e-05, + "loss": 7.406627264572307e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.01870773173868656, + "learning_rate": 3.11727834939056e-05, + "loss": 0.0001340095914201811, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00013, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.08653457462787628, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0008557374821975827, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00086, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.49505236744880676, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.007139122113585472, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00716, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.033166300505399704, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00024455596576444805, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00024, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.09, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.026151562109589577, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00018419645493850112, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00018, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.5249567627906799, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.03347098454833031, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03404, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.008504888974130154, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00012848488404415548, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00013, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.08, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.006400417070835829, + "learning_rate": 2.9473853553877484e-05, + "loss": 7.728980563115329e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.010285867378115654, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00018826585437636822, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.05493743345141411, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0002043755230261013, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.00775935361161828, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00012459162098821253, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.21, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.05490761622786522, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.000744640186894685, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.006251805927604437, + "learning_rate": 2.829199644117484e-05, + "loss": 0.00010329978249501437, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0001, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.01030012033879757, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001805450883693993, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.007539911661297083, + "learning_rate": 2.782696506053033e-05, + "loss": 0.00016556130140088499, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00017, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.004288971424102783, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.0001009388652164489, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.17, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.01721479743719101, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0003073053085245192, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00031, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.008515549823641777, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00017193684470839798, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00017, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.009272439405322075, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00021900574211031199, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00022, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.034540578722953796, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0005546602769754827, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00055, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.013745410367846489, + "learning_rate": 2.645931522709877e-05, + "loss": 0.0003212787851225585, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00032, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.005436223931610584, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001245749299414456, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00012, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.010458163917064667, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0002482631243765354, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00025, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.006481671240180731, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0001577583752805367, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.012174229137599468, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00027598816086538136, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00028, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.011468137614428997, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00023577807587571442, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00024, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.005027628969401121, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.00010462553473189473, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.005808121990412474, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00013498810585588217, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00013, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0358550138771534, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.0006072560790926218, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00061, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.09168751537799835, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0007024870719760656, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0007, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.35, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.014628843404352665, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00020378318731673062, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.05326993018388748, + "learning_rate": 2.406442693028651e-05, + "loss": 0.00035949741140939295, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00036, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.008257771842181683, + "learning_rate": 2.3854255584458547e-05, + "loss": 9.612501162337139e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.03791302442550659, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0004905672976747155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00049, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2622528076171875, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.003920107148587704, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00393, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.06083231046795845, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0009291216847486794, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00093, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.014715000987052917, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00020523657440207899, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00021, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.01740424521267414, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00030176169821061194, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.0003, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.005717985797673464, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00011773040023399517, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00012, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005951862782239914, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.0001332889514742419, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00013, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.0560171976685524, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00048283624346368015, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.013027072884142399, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00022208130394574255, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00022, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.0371403768658638, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0005342152435332537, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00053, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.68, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.004872396122664213, + "learning_rate": 2.1629798596457056e-05, + "loss": 7.259240373969078e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00007, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.010584473609924316, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.00017825220129452646, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00018, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.007694529369473457, + "learning_rate": 2.124308220868431e-05, + "loss": 7.509582792408764e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00008, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.02000581845641136, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0002455090289004147, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.0213518887758255, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.0002898501115851104, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00029, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004943408537656069, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0001215632728417404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00012, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.39, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.012947587296366692, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001342954346910119, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00013, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.05408314988017082, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00036081415601074696, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.037419769912958145, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00042080413550138474, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00042, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.035673510283231735, + "learning_rate": 1.993423839463052e-05, + "loss": 0.000293174380203709, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00029, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0039419797249138355, + "learning_rate": 1.975303621298445e-05, + "loss": 8.023489499464631e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00008, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.05340631678700447, + "learning_rate": 1.957330080137385e-05, + "loss": 0.00029110765899531543, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.016526181250810623, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00015069378423504531, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.008226566947996616, + "learning_rate": 1.9218260145006073e-05, + "loss": 9.167128155240789e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00009, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.005769859533756971, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00011197607091162354, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00011, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.08041926473379135, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.000366955588106066, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00037, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.03303743153810501, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0001731862430460751, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.008530828170478344, + "learning_rate": 1.85261050441233e-05, + "loss": 0.00011680425814120099, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00012, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 190216 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.0015226984396576881, + "learning_rate": 1.8356842992397304e-05, + "loss": 3.2423220545751974e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00003, + "step": 417, + "tokens/total": 12617168, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 190708 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.03927593678236008, + "learning_rate": 1.8189105812005714e-05, + "loss": 0.00023564108414575458, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 418, + "tokens/total": 12647680, + "tokens/train_per_sec_per_gpu": 31.01, + "tokens/trainable": 191126 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.06375617533922195, + "learning_rate": 1.802290048317732e-05, + "loss": 0.0004688606131821871, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00047, + "step": 419, + "tokens/total": 12677952, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 191540 + }, + { + "epoch": 1.640625, + "grad_norm": 0.01704459637403488, + "learning_rate": 1.785823392239424e-05, + "loss": 0.00010123683023266494, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 420, + "tokens/total": 12708416, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.0012392516946420074, + "learning_rate": 1.7695112982104225e-05, + "loss": 2.009748641285114e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 421, + "tokens/total": 12738640, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 192463 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.06435113400220871, + "learning_rate": 1.7533544450435433e-05, + "loss": 0.0006201984360814095, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00062, + "step": 422, + "tokens/total": 12769232, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 192918 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.04247846081852913, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.0001703215966699645, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00017, + "step": 423, + "tokens/total": 12799648, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 193425 + }, + { + "epoch": 1.65625, + "grad_norm": 0.0037633716128766537, + "learning_rate": 1.721509144218405e-05, + "loss": 7.026562525425106e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00007, + "step": 424, + "tokens/total": 12829920, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 193898 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.24774393439292908, + "learning_rate": 1.705822021773101e-05, + "loss": 0.0018032332882285118, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0018, + "step": 425, + "tokens/total": 12860112, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 194315 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.0021040684077888727, + "learning_rate": 1.69029279056068e-05, + "loss": 3.659192589111626e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00004, + "step": 426, + "tokens/total": 12890320, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 194796 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.001372064813040197, + "learning_rate": 1.6749220968158415e-05, + "loss": 2.2382890165317804e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00002, + "step": 427, + "tokens/total": 12920656, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 195262 + }, + { + "epoch": 1.671875, + "grad_norm": 0.004309098701924086, + "learning_rate": 1.659710580175893e-05, + "loss": 5.155909457243979e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 428, + "tokens/total": 12951040, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 195680 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.03517591208219528, + "learning_rate": 1.644658873654133e-05, + "loss": 0.00021132534311618656, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00021, + "step": 429, + "tokens/total": 12981232, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 196145 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.005870590452104807, + "learning_rate": 1.629767603613508e-05, + "loss": 6.348404713207856e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00006, + "step": 430, + "tokens/total": 13011616, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 196626 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.2625119984149933, + "learning_rate": 1.615037389740547e-05, + "loss": 0.011768318712711334, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01184, + "step": 431, + "tokens/total": 13042208, + "tokens/train_per_sec_per_gpu": 31.22, + "tokens/trainable": 197067 + }, + { + "epoch": 1.6875, + "grad_norm": 0.007063967175781727, + "learning_rate": 1.600468845019576e-05, + "loss": 4.784313205163926e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00005, + "step": 432, + "tokens/total": 13072800, + "tokens/train_per_sec_per_gpu": 38.79, + "tokens/trainable": 197565 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.006292980629950762, + "learning_rate": 1.5860625757072092e-05, + "loss": 5.077052628621459e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00005, + "step": 433, + "tokens/total": 13103328, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 198015 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.0020431573502719402, + "learning_rate": 1.571819181307116e-05, + "loss": 3.991067933384329e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00004, + "step": 434, + "tokens/total": 13133472, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 198442 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.004255881067365408, + "learning_rate": 1.557739254545075e-05, + "loss": 5.153669189894572e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00005, + "step": 435, + "tokens/total": 13164064, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 198914 + }, + { + "epoch": 1.703125, + "grad_norm": 0.46879518032073975, + "learning_rate": 1.543823381344311e-05, + "loss": 0.013075481168925762, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01316, + "step": 436, + "tokens/total": 13194464, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 199355 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.22727783024311066, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.0031720949336886406, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00318, + "step": 437, + "tokens/total": 13224992, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 199830 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.003598392242565751, + "learning_rate": 1.5164861051607254e-05, + "loss": 5.355676694307476e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00005, + "step": 438, + "tokens/total": 13255296, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 200338 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.04593397676944733, + "learning_rate": 1.5030658397935521e-05, + "loss": 0.00022313687077257782, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00022, + "step": 439, + "tokens/total": 13285344, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 200785 + }, + { + "epoch": 1.71875, + "grad_norm": 0.0011655986309051514, + "learning_rate": 1.4898119031716104e-05, + "loss": 2.53522812272422e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 440, + "tokens/total": 13315632, + "tokens/train_per_sec_per_gpu": 36.4, + "tokens/trainable": 201267 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.010816111229360104, + "learning_rate": 1.476724846845306e-05, + "loss": 0.0002002313849516213, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0002, + "step": 441, + "tokens/total": 13346048, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 201734 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.008938372135162354, + "learning_rate": 1.463805215420471e-05, + "loss": 0.00011111146159237251, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 442, + "tokens/total": 13376304, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 202174 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.010652483440935612, + "learning_rate": 1.451053546535705e-05, + "loss": 0.000100844117696397, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 443, + "tokens/total": 13406720, + "tokens/train_per_sec_per_gpu": 36.34, + "tokens/trainable": 202612 + }, + { + "epoch": 1.734375, + "grad_norm": 0.005689944606274366, + "learning_rate": 1.438470370840001e-05, + "loss": 0.00011003226973116398, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00011, + "step": 444, + "tokens/total": 13436464, + "tokens/train_per_sec_per_gpu": 31.09, + "tokens/trainable": 203020 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.24016952514648438, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.002352698938921094, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00236, + "step": 445, + "tokens/total": 13466672, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 203479 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.008882527239620686, + "learning_rate": 1.413811586531508e-05, + "loss": 0.0001598334638401866, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00016, + "step": 446, + "tokens/total": 13496736, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 203946 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.005755060818046331, + "learning_rate": 1.4017370040713884e-05, + "loss": 7.665179145988077e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 447, + "tokens/total": 13526928, + "tokens/train_per_sec_per_gpu": 37.64, + "tokens/trainable": 204382 + }, + { + "epoch": 1.75, + "grad_norm": 0.003990499302744865, + "learning_rate": 1.3898329670629645e-05, + "loss": 8.206715574488044e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00008, + "step": 448, + "tokens/total": 13557392, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 204830 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.0017717559821903706, + "learning_rate": 1.3780999708818058e-05, + "loss": 3.886378544848412e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00004, + "step": 449, + "tokens/total": 13587776, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 205300 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.004407316446304321, + "learning_rate": 1.3665385037857758e-05, + "loss": 6.286771531449631e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 450, + "tokens/total": 13618112, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 205726 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.004234767984598875, + "learning_rate": 1.3551490468947126e-05, + "loss": 0.00010631309851305559, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00011, + "step": 451, + "tokens/total": 13648512, + "tokens/train_per_sec_per_gpu": 31.77, + "tokens/trainable": 206163 + }, + { + "epoch": 1.765625, + "grad_norm": 0.19428566098213196, + "learning_rate": 1.3439320741704075e-05, + "loss": 0.008113588206470013, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00815, + "step": 452, + "tokens/total": 13678704, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 206619 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.35631781816482544, + "learning_rate": 1.3328880523968808e-05, + "loss": 0.004253438673913479, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00426, + "step": 453, + "tokens/total": 13709232, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 207101 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.004516016226261854, + "learning_rate": 1.3220174411609587e-05, + "loss": 5.911413245485164e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 454, + "tokens/total": 13739600, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 207546 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.007509376388043165, + "learning_rate": 1.3113206928331471e-05, + "loss": 0.00010886647214647382, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00011, + "step": 455, + "tokens/total": 13769936, + "tokens/train_per_sec_per_gpu": 34.43, + "tokens/trainable": 207989 + }, + { + "epoch": 1.78125, + "grad_norm": 0.0016118305502459407, + "learning_rate": 1.300798252548806e-05, + "loss": 3.1790965294931084e-05, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00003, + "step": 456, + "tokens/total": 13800240, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 208402 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.0032239535357803106, + "learning_rate": 1.2904505581896265e-05, + "loss": 4.2626015783753246e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00004, + "step": 457, + "tokens/total": 13830512, + "tokens/train_per_sec_per_gpu": 36.74, + "tokens/trainable": 208904 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.0048893350176513195, + "learning_rate": 1.2802780403654082e-05, + "loss": 6.955669960007071e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 458, + "tokens/total": 13860832, + "tokens/train_per_sec_per_gpu": 33.78, + "tokens/trainable": 209382 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.003503630170598626, + "learning_rate": 1.2702811223961408e-05, + "loss": 7.862070197006688e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00008, + "step": 459, + "tokens/total": 13890992, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 209833 + }, + { + "epoch": 1.796875, + "grad_norm": 0.34274882078170776, + "learning_rate": 1.2604602202943861e-05, + "loss": 0.010280012153089046, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01033, + "step": 460, + "tokens/total": 13921424, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 210284 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.15473094582557678, + "learning_rate": 1.2508157427479686e-05, + "loss": 0.0024835069198161364, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00249, + "step": 461, + "tokens/total": 13951504, + "tokens/train_per_sec_per_gpu": 34.19, + "tokens/trainable": 210769 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.0028502352070063353, + "learning_rate": 1.2413480911029655e-05, + "loss": 5.3788880904903635e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00005, + "step": 462, + "tokens/total": 13979728, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 211193 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.0012627877295017242, + "learning_rate": 1.2320576593470082e-05, + "loss": 3.094038038398139e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00003, + "step": 463, + "tokens/total": 14009936, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 211655 + }, + { + "epoch": 1.8125, + "grad_norm": 0.0036251619458198547, + "learning_rate": 1.2229448340928828e-05, + "loss": 8.078858081717044e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00008, + "step": 464, + "tokens/total": 14039872, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 212085 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.017260659486055374, + "learning_rate": 1.2140099945624458e-05, + "loss": 0.0002048999012913555, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 465, + "tokens/total": 14070096, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 212570 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.2109188288450241, + "learning_rate": 1.205253512570841e-05, + "loss": 0.006965605076402426, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00699, + "step": 466, + "tokens/total": 14100192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 213011 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.052500564604997635, + "learning_rate": 1.1966757525110255e-05, + "loss": 0.0004830099060200155, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00048, + "step": 467, + "tokens/total": 14130448, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 213450 + }, + { + "epoch": 1.828125, + "grad_norm": 0.005511680152267218, + "learning_rate": 1.1882770713386095e-05, + "loss": 8.27932235551998e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00008, + "step": 468, + "tokens/total": 14160480, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 213898 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.14864490926265717, + "learning_rate": 1.180057818556998e-05, + "loss": 0.0016087992116808891, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00161, + "step": 469, + "tokens/total": 14191152, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 214366 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.004650567192584276, + "learning_rate": 1.1720183362028494e-05, + "loss": 9.897982818074524e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 470, + "tokens/total": 14219392, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 214800 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.06575815379619598, + "learning_rate": 1.1641589588318387e-05, + "loss": 0.0007422211347147822, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 471, + "tokens/total": 14249920, + "tokens/train_per_sec_per_gpu": 35.59, + "tokens/trainable": 215292 + }, + { + "epoch": 1.84375, + "grad_norm": 0.002274462953209877, + "learning_rate": 1.1564800135047418e-05, + "loss": 5.863649130333215e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00006, + "step": 472, + "tokens/total": 14280048, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 215741 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.002968342276290059, + "learning_rate": 1.148981819773816e-05, + "loss": 6.629896233789623e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 473, + "tokens/total": 14310432, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 216194 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.004850646015256643, + "learning_rate": 1.1416646896695086e-05, + "loss": 9.518570732325315e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0001, + "step": 474, + "tokens/total": 14340496, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 216634 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.023391559720039368, + "learning_rate": 1.1345289276874717e-05, + "loss": 0.00047331867972388864, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 475, + "tokens/total": 14370720, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 217040 + }, + { + "epoch": 1.859375, + "grad_norm": 0.00271675456315279, + "learning_rate": 1.1275748307758873e-05, + "loss": 5.663794945576228e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 476, + "tokens/total": 14401136, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 217530 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.018746392801404, + "learning_rate": 1.1208026883231147e-05, + "loss": 0.00019438326125964522, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00019, + "step": 477, + "tokens/total": 14431360, + "tokens/train_per_sec_per_gpu": 34.65, + "tokens/trainable": 217988 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.12623320519924164, + "learning_rate": 1.1142127821456433e-05, + "loss": 0.0016139973886311054, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00162, + "step": 478, + "tokens/total": 14461968, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 218461 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.00717399176210165, + "learning_rate": 1.1078053864763674e-05, + "loss": 0.00011526358139235526, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 479, + "tokens/total": 14491984, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 218909 + }, + { + "epoch": 1.875, + "grad_norm": 0.05601891130208969, + "learning_rate": 1.1015807679531756e-05, + "loss": 0.0003784724394790828, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00038, + "step": 480, + "tokens/total": 14522544, + "tokens/train_per_sec_per_gpu": 34.19, + "tokens/trainable": 219350 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.857299273093647e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-480/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..240bff6161f308f23188ba6ced3f99ca3a7d0813 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:777863e8bd407287f6fc4ae90c64cd69b9931e7cafe53c7e9a27afc677d964a1 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a88b2f0207db3a60d0a743eedbd37a0b3d33bf89 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b16c11fd1c653a7d58e19028bf3fdcd0c56296cd0af05df28cbfd395fe2e0e01 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..93906269157a5485207c9bb4b3b757d5ea700afa --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:75e98dcceb68f1b6ee34a96307625206231cf5a0e0e4e16d8a9eafccc33ea3fe +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..22325b3b3d08e0697784fe21b8c321485a51f7b5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..242dd434cbe768e1e6c671a04c0c09f21072a0fc --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokens_state.json @@ -0,0 +1 @@ +{"total": 15491760, "trainable": 234078} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..25c06b1e7c5e92f836b6609cb54aee2cf8fe64fa --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/trainer_state.json @@ -0,0 +1,7202 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 512, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.42756563425064087, + "learning_rate": 9.53619478457953e-05, + "loss": 0.004547779448330402, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00456, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.9736410975456238, + "learning_rate": 9.523275153154695e-05, + "loss": 0.04175024852156639, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04263, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.88, + "tokens/trainable": 44739 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.8381142020225525, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013025594875216484, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01311, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 45202 + }, + { + "epoch": 0.390625, + "grad_norm": 0.22207525372505188, + "learning_rate": 9.49693416020645e-05, + "loss": 0.004689273424446583, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0047, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 45646 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.28477466106414795, + "learning_rate": 9.483513894839276e-05, + "loss": 0.010324095375835896, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01038, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 46089 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.5098783373832703, + "learning_rate": 9.469927859198888e-05, + "loss": 0.028538472950458527, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02895, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 46547 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.3906193673610687, + "learning_rate": 9.456176618655689e-05, + "loss": 0.017888592556118965, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01805, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 47014 + }, + { + "epoch": 0.40625, + "grad_norm": 0.5902650952339172, + "learning_rate": 9.442260745454927e-05, + "loss": 0.03132036700844765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.03182, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 47503 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.2547222375869751, + "learning_rate": 9.428180818692884e-05, + "loss": 0.014553382061421871, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01466, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.37, + "tokens/trainable": 47996 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.49491167068481445, + "learning_rate": 9.413937424292791e-05, + "loss": 0.0295359306037426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02998, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.18, + "tokens/trainable": 48447 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.481608510017395, + "learning_rate": 9.399531154980424e-05, + "loss": 0.017734361812472343, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01789, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.78, + "tokens/trainable": 48889 + }, + { + "epoch": 0.421875, + "grad_norm": 0.3913033604621887, + "learning_rate": 9.384962610259455e-05, + "loss": 0.022145511582493782, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02239, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 49360 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.1751519739627838, + "learning_rate": 9.370232396386494e-05, + "loss": 0.0074960459023714066, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00752, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 49827 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.28872016072273254, + "learning_rate": 9.355341126345868e-05, + "loss": 0.01847866177558899, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01865, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 50267 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3326593041419983, + "learning_rate": 9.340289419824107e-05, + "loss": 0.013967925682663918, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01407, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 50711 + }, + { + "epoch": 0.4375, + "grad_norm": 0.47660934925079346, + "learning_rate": 9.325077903184159e-05, + "loss": 0.02260858193039894, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02287, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 51157 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.3481523096561432, + "learning_rate": 9.30970720943932e-05, + "loss": 0.008511725813150406, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00855, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 51613 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.4288201928138733, + "learning_rate": 9.2941779782269e-05, + "loss": 0.02450375072658062, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02481, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 52017 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.17284883558750153, + "learning_rate": 9.278490855781596e-05, + "loss": 0.007882107980549335, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00791, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 52506 + }, + { + "epoch": 0.453125, + "grad_norm": 0.2693459689617157, + "learning_rate": 9.262646494908604e-05, + "loss": 0.005466018337756395, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00548, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.77, + "tokens/trainable": 53008 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.48690056800842285, + "learning_rate": 9.246645554956457e-05, + "loss": 0.004998547025024891, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00501, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 53467 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.43558549880981445, + "learning_rate": 9.230488701789578e-05, + "loss": 0.017219291999936104, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01737, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 53923 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.28566548228263855, + "learning_rate": 9.214176607760577e-05, + "loss": 0.00544874370098114, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00546, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 54363 + }, + { + "epoch": 0.46875, + "grad_norm": 0.07611701637506485, + "learning_rate": 9.197709951682268e-05, + "loss": 0.0014362478395923972, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00144, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 54839 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.13201506435871124, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0024022101424634457, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00241, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 55267 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.40975725650787354, + "learning_rate": 9.164315700760271e-05, + "loss": 0.008171994239091873, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00821, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 55700 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.15150536596775055, + "learning_rate": 9.147389495587671e-05, + "loss": 0.0036906469613313675, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.0037, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 56144 + }, + { + "epoch": 0.484375, + "grad_norm": 0.514573335647583, + "learning_rate": 9.130311507650116e-05, + "loss": 0.01570757105946541, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01583, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 56589 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.43223315477371216, + "learning_rate": 9.113082447632394e-05, + "loss": 0.007733152247965336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 57081 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.2972177565097809, + "learning_rate": 9.09570303250602e-05, + "loss": 0.008276824839413166, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00831, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 57527 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.6973468065261841, + "learning_rate": 9.078173985499394e-05, + "loss": 0.011385848745703697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01145, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 58005 + }, + { + "epoch": 0.5, + "grad_norm": 0.11440204083919525, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0026649129576981068, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00267, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 58454 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.4737952649593353, + "learning_rate": 9.042669919862615e-05, + "loss": 0.011923262849450111, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.01199, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.11, + "tokens/trainable": 58896 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.22904613614082336, + "learning_rate": 9.024696378701557e-05, + "loss": 0.003364444011822343, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00337, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 59356 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.6996368169784546, + "learning_rate": 9.006576160536948e-05, + "loss": 0.0145557951182127, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01466, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 59792 + }, + { + "epoch": 0.515625, + "grad_norm": 0.18643876910209656, + "learning_rate": 8.988310019425035e-05, + "loss": 0.004168116021901369, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00418, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 60266 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.3754265606403351, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01239765528589487, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01247, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 60759 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.5720171332359314, + "learning_rate": 8.951343014914869e-05, + "loss": 0.028643302619457245, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02906, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.88, + "tokens/trainable": 61204 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.21384330093860626, + "learning_rate": 8.932643689864568e-05, + "loss": 0.005144410766661167, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00516, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 61684 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3025348484516144, + "learning_rate": 8.913801518498845e-05, + "loss": 0.011437858454883099, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.0115, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.36, + "tokens/trainable": 62159 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.16174718737602234, + "learning_rate": 8.894817284917364e-05, + "loss": 0.005412050988525152, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00543, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 62649 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.10317311435937881, + "learning_rate": 8.875691779131569e-05, + "loss": 0.002703815931454301, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00271, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 63100 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.23210155963897705, + "learning_rate": 8.856425797031829e-05, + "loss": 0.00892355665564537, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00896, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 63588 + }, + { + "epoch": 0.546875, + "grad_norm": 0.4230027496814728, + "learning_rate": 8.837020140354295e-05, + "loss": 0.026184625923633575, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02653, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 64005 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.5172422528266907, + "learning_rate": 8.817475616647554e-05, + "loss": 0.004686606116592884, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0047, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 64462 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.28759023547172546, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009430551901459694, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00948, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 64900 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.45416516065597534, + "learning_rate": 8.777973227201069e-05, + "loss": 0.025679660961031914, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.02601, + "step": 143, + "tokens/total": 4323248, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 65379 + }, + { + "epoch": 0.5625, + "grad_norm": 0.09341549873352051, + "learning_rate": 8.758017005316988e-05, + "loss": 0.0024609356187283993, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00246, + "step": 144, + "tokens/total": 4353344, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 65845 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.06811556965112686, + "learning_rate": 8.737925204046629e-05, + "loss": 0.00176865397952497, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00177, + "step": 145, + "tokens/total": 4383584, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 66285 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.3615569472312927, + "learning_rate": 8.717698659491851e-05, + "loss": 0.018001921474933624, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01816, + "step": 146, + "tokens/total": 4413856, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 66736 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.37582314014434814, + "learning_rate": 8.697338213361735e-05, + "loss": 0.01264512725174427, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01273, + "step": 147, + "tokens/total": 4444112, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 67150 + }, + { + "epoch": 0.578125, + "grad_norm": 0.3228939473628998, + "learning_rate": 8.676844712937552e-05, + "loss": 0.009639455936849117, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00969, + "step": 148, + "tokens/total": 4474464, + "tokens/train_per_sec_per_gpu": 35.2, + "tokens/trainable": 67633 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.44322797656059265, + "learning_rate": 8.656219011037509e-05, + "loss": 0.013246036134660244, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01333, + "step": 149, + "tokens/total": 4504880, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 68077 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.27842915058135986, + "learning_rate": 8.63546196598125e-05, + "loss": 0.007822944782674313, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00785, + "step": 150, + "tokens/total": 4534976, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 68545 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.20541682839393616, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005034038331359625, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00505, + "step": 151, + "tokens/total": 4565088, + "tokens/train_per_sec_per_gpu": 30.33, + "tokens/trainable": 69011 + }, + { + "epoch": 0.59375, + "grad_norm": 0.33200570940971375, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007599984761327505, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00763, + "step": 152, + "tokens/total": 4595600, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 69465 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.47942468523979187, + "learning_rate": 8.572411436841618e-05, + "loss": 0.011957819573581219, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01203, + "step": 153, + "tokens/total": 4626032, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 69928 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.13327482342720032, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0018434002995491028, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00185, + "step": 154, + "tokens/total": 4656352, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 70395 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.2294115424156189, + "learning_rate": 8.529737015125824e-05, + "loss": 0.003923698328435421, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00393, + "step": 155, + "tokens/total": 4684672, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 70829 + }, + { + "epoch": 0.609375, + "grad_norm": 0.18192212283611298, + "learning_rate": 8.508210239396639e-05, + "loss": 0.0024503993336111307, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00245, + "step": 156, + "tokens/total": 4715216, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 71287 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.015309900976717472, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0003224749234504998, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00032, + "step": 157, + "tokens/total": 4745376, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 71749 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.39558145403862, + "learning_rate": 8.464782037243449e-05, + "loss": 0.02732035145163536, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0277, + "step": 158, + "tokens/total": 4775776, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72185 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.2731363773345947, + "learning_rate": 8.442882418044202e-05, + "loss": 0.00566218001767993, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00568, + "step": 159, + "tokens/total": 4805856, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 72604 + }, + { + "epoch": 0.625, + "grad_norm": 0.41349318623542786, + "learning_rate": 8.420860333495179e-05, + "loss": 0.00760249700397253, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00763, + "step": 160, + "tokens/total": 4836304, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 73061 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.1329299956560135, + "learning_rate": 8.398716700025208e-05, + "loss": 0.002723761135712266, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00273, + "step": 161, + "tokens/total": 4866544, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 73474 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.2733931839466095, + "learning_rate": 8.376452439121266e-05, + "loss": 0.008426842279732227, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00846, + "step": 162, + "tokens/total": 4897120, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 73923 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.6801484227180481, + "learning_rate": 8.354068477290124e-05, + "loss": 0.02508145570755005, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0254, + "step": 163, + "tokens/total": 4927440, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 74379 + }, + { + "epoch": 0.640625, + "grad_norm": 0.22900699079036713, + "learning_rate": 8.331565746019807e-05, + "loss": 0.006623784080147743, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00665, + "step": 164, + "tokens/total": 4958032, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 74830 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.1291102021932602, + "learning_rate": 8.308945181740812e-05, + "loss": 0.0024572773836553097, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00246, + "step": 165, + "tokens/total": 4988368, + "tokens/train_per_sec_per_gpu": 31.84, + "tokens/trainable": 75262 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.12582950294017792, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0014331504935398698, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00143, + "step": 166, + "tokens/total": 5018672, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 75703 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.10334278643131256, + "learning_rate": 8.263354324357182e-05, + "loss": 0.003322131698951125, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00333, + "step": 167, + "tokens/total": 5049344, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 76164 + }, + { + "epoch": 0.65625, + "grad_norm": 0.344831258058548, + "learning_rate": 8.240385928474219e-05, + "loss": 0.01979166269302368, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01999, + "step": 168, + "tokens/total": 5079856, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 76623 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.3123904764652252, + "learning_rate": 8.217303493946967e-05, + "loss": 0.020188894122838974, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02039, + "step": 169, + "tokens/total": 5110192, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 77047 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.24240995943546295, + "learning_rate": 8.194107981329746e-05, + "loss": 0.01431284286081791, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01442, + "step": 170, + "tokens/total": 5140592, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 77496 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.25977423787117004, + "learning_rate": 8.170800355882518e-05, + "loss": 0.014354058541357517, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01446, + "step": 171, + "tokens/total": 5170848, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 77956 + }, + { + "epoch": 0.671875, + "grad_norm": 0.19354988634586334, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004821081180125475, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00483, + "step": 172, + "tokens/total": 5201280, + "tokens/train_per_sec_per_gpu": 29.87, + "tokens/trainable": 78410 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.2551465928554535, + "learning_rate": 8.123852650824877e-05, + "loss": 0.010946989059448242, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01101, + "step": 173, + "tokens/total": 5231712, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 78849 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.34295886754989624, + "learning_rate": 8.100214524900103e-05, + "loss": 0.02883588895201683, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.02926, + "step": 174, + "tokens/total": 5261440, + "tokens/train_per_sec_per_gpu": 30.2, + "tokens/trainable": 79266 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.0546368770301342, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0018831162014976144, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00188, + "step": 175, + "tokens/total": 5291728, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 79736 + }, + { + "epoch": 0.6875, + "grad_norm": 0.22756348550319672, + "learning_rate": 8.052614644612253e-05, + "loss": 0.0047474936582148075, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00476, + "step": 176, + "tokens/total": 5322240, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 80214 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.06755488365888596, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0017190770013257861, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00172, + "step": 177, + "tokens/total": 5352576, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 80661 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.2727944850921631, + "learning_rate": 8.004589869885986e-05, + "loss": 0.00690429238602519, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00693, + "step": 178, + "tokens/total": 5382784, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 81129 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.20468388497829437, + "learning_rate": 7.980420642489674e-05, + "loss": 0.003946482669562101, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00395, + "step": 179, + "tokens/total": 5412784, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 81576 + }, + { + "epoch": 0.703125, + "grad_norm": 0.25605547428131104, + "learning_rate": 7.95614819466576e-05, + "loss": 0.004923749715089798, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00494, + "step": 180, + "tokens/total": 5442960, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 82049 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.33550992608070374, + "learning_rate": 7.931773536489872e-05, + "loss": 0.010120240971446037, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01017, + "step": 181, + "tokens/total": 5473216, + "tokens/train_per_sec_per_gpu": 38.84, + "tokens/trainable": 82566 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.22376962006092072, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025828760117292404, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00259, + "step": 182, + "tokens/total": 5503568, + "tokens/train_per_sec_per_gpu": 33.98, + "tokens/trainable": 83054 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.023590704426169395, + "learning_rate": 7.882721650609442e-05, + "loss": 0.0007615481736138463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00076, + "step": 183, + "tokens/total": 5533760, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 83501 + }, + { + "epoch": 0.71875, + "grad_norm": 0.06294015794992447, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0010393422562628984, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00104, + "step": 184, + "tokens/total": 5564240, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 84007 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.04545356705784798, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0006121923215687275, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00061, + "step": 185, + "tokens/total": 5594448, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 84491 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.23753786087036133, + "learning_rate": 7.808402738346527e-05, + "loss": 0.00197276147082448, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00197, + "step": 186, + "tokens/total": 5624816, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 84952 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.12784555554389954, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0010586264543235302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 187, + "tokens/total": 5655104, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 85400 + }, + { + "epoch": 0.734375, + "grad_norm": 0.8053486347198486, + "learning_rate": 7.758374768294647e-05, + "loss": 0.01503228023648262, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01515, + "step": 188, + "tokens/total": 5685360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 85889 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6335012316703796, + "learning_rate": 7.733219291524489e-05, + "loss": 0.004618915729224682, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00463, + "step": 189, + "tokens/total": 5715648, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 86396 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.7543070912361145, + "learning_rate": 7.707970881383977e-05, + "loss": 0.03686961531639099, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.03756, + "step": 190, + "tokens/total": 5746240, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 86867 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.5514787435531616, + "learning_rate": 7.682630588562518e-05, + "loss": 0.023130130022764206, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0234, + "step": 191, + "tokens/total": 5776816, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 87314 + }, + { + "epoch": 0.75, + "grad_norm": 0.30548200011253357, + "learning_rate": 7.657199467573129e-05, + "loss": 0.002730845008045435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00273, + "step": 192, + "tokens/total": 5807248, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 87736 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.42851904034614563, + "learning_rate": 7.631678576708561e-05, + "loss": 0.015687599778175354, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01581, + "step": 193, + "tokens/total": 5837680, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 88219 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.14844387769699097, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0037631455343216658, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00377, + "step": 194, + "tokens/total": 5868000, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 88699 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.19134196639060974, + "learning_rate": 7.580371737159148e-05, + "loss": 0.004276394844055176, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00429, + "step": 195, + "tokens/total": 5896224, + "tokens/train_per_sec_per_gpu": 30.51, + "tokens/trainable": 89132 + }, + { + "epoch": 0.765625, + "grad_norm": 0.44769415259361267, + "learning_rate": 7.554587923561324e-05, + "loss": 0.013675286434590816, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01377, + "step": 196, + "tokens/total": 5926624, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 89594 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11093451082706451, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0035221045836806297, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00353, + "step": 197, + "tokens/total": 5954944, + "tokens/train_per_sec_per_gpu": 38.22, + "tokens/trainable": 90042 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.19566713273525238, + "learning_rate": 7.502764873523431e-05, + "loss": 0.00824933685362339, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00828, + "step": 198, + "tokens/total": 5985200, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 90503 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.1111554354429245, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002672867150977254, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00268, + "step": 199, + "tokens/total": 6015952, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 90982 + }, + { + "epoch": 0.78125, + "grad_norm": 0.2508006989955902, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0060717021115124226, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00609, + "step": 200, + "tokens/total": 6046208, + "tokens/train_per_sec_per_gpu": 39.5, + "tokens/trainable": 91466 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.07083698362112045, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0025744480080902576, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00258, + "step": 201, + "tokens/total": 6076208, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 91912 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.06263043731451035, + "learning_rate": 7.398127346871986e-05, + "loss": 0.00145058985799551, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00145, + "step": 202, + "tokens/total": 6106560, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 92385 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.44214949011802673, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02974056825041771, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03019, + "step": 203, + "tokens/total": 6136640, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 92842 + }, + { + "epoch": 0.796875, + "grad_norm": 0.39396584033966064, + "learning_rate": 7.345330287655617e-05, + "loss": 0.011925775557756424, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.012, + "step": 204, + "tokens/total": 6167184, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 93323 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.5901291370391846, + "learning_rate": 7.31881602037339e-05, + "loss": 0.019959867000579834, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.02016, + "step": 205, + "tokens/total": 6197376, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 93756 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.2788645327091217, + "learning_rate": 7.29222606473245e-05, + "loss": 0.006857238709926605, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00688, + "step": 206, + "tokens/total": 6227600, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 94235 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.19782070815563202, + "learning_rate": 7.265561527249383e-05, + "loss": 0.005047307349741459, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00506, + "step": 207, + "tokens/total": 6257760, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 94733 + }, + { + "epoch": 0.8125, + "grad_norm": 0.14907793700695038, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0029104873538017273, + "memory/device_reserved (GiB)": 35.93, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00291, + "step": 208, + "tokens/total": 6288048, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 95218 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.03554432839155197, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0010063875233754516, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00101, + "step": 209, + "tokens/total": 6318688, + "tokens/train_per_sec_per_gpu": 29.31, + "tokens/trainable": 95619 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.11122506856918335, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0032706214115023613, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00328, + "step": 210, + "tokens/total": 6348944, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 96076 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.38507047295570374, + "learning_rate": 7.158179796885005e-05, + "loss": 0.006026511080563068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00604, + "step": 211, + "tokens/total": 6378960, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 96505 + }, + { + "epoch": 0.828125, + "grad_norm": 0.5194666385650635, + "learning_rate": 7.131159054949273e-05, + "loss": 0.008640932850539684, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00868, + "step": 212, + "tokens/total": 6409472, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 96930 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.0659736841917038, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0014961843844503164, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0015, + "step": 213, + "tokens/total": 6439904, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 97361 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.233027845621109, + "learning_rate": 7.076915060786705e-05, + "loss": 0.00338245602324605, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00339, + "step": 214, + "tokens/total": 6470176, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 97858 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.38389813899993896, + "learning_rate": 7.049694065873882e-05, + "loss": 0.014850301668047905, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01496, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 98308 + }, + { + "epoch": 0.84375, + "grad_norm": 0.07625724375247955, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0009395676897838712, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00094, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 98781 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.1813930720090866, + "learning_rate": 6.99505974422157e-05, + "loss": 0.005376772489398718, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00539, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 99266 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.28488755226135254, + "learning_rate": 6.967648691039213e-05, + "loss": 0.007393942214548588, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00742, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.74, + "tokens/trainable": 99731 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.2636983394622803, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0031264584977179766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00313, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 100164 + }, + { + "epoch": 0.859375, + "grad_norm": 0.19989553093910217, + "learning_rate": 6.912644503343682e-05, + "loss": 0.018466733396053314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01864, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.33, + "tokens/trainable": 100620 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.1063912957906723, + "learning_rate": 6.885053657779273e-05, + "loss": 0.0018869235645979643, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00189, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 101080 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.41694170236587524, + "learning_rate": 6.857405174478604e-05, + "loss": 0.025706909596920013, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02604, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.93, + "tokens/trainable": 101546 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.7918646335601807, + "learning_rate": 6.82970020400792e-05, + "loss": 0.01185107696801424, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01192, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.49, + "tokens/trainable": 101987 + }, + { + "epoch": 0.875, + "grad_norm": 0.32625025510787964, + "learning_rate": 6.801939899284132e-05, + "loss": 0.009739323519170284, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00979, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 102435 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.4342412054538727, + "learning_rate": 6.774125415526827e-05, + "loss": 0.009223651140928268, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00927, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 102843 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.13243624567985535, + "learning_rate": 6.746257910210214e-05, + "loss": 0.0032694521360099316, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00327, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103342 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.09638779610395432, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0033315662294626236, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00334, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.85, + "tokens/trainable": 103844 + }, + { + "epoch": 0.890625, + "grad_norm": 0.3315463960170746, + "learning_rate": 6.69036847577983e-05, + "loss": 0.020162900909781456, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.02037, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 104307 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.23732618987560272, + "learning_rate": 6.662348872453553e-05, + "loss": 0.014629160054028034, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.01474, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 104772 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.06660830229520798, + "learning_rate": 6.63428089904618e-05, + "loss": 0.0017317197052761912, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00173, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105240 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.4719278812408447, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00895722210407257, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.009, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 105703 + }, + { + "epoch": 0.90625, + "grad_norm": 0.18451477587223053, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0027692944277077913, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00277, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 106179 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.07695464044809341, + "learning_rate": 6.549798448339548e-05, + "loss": 0.002346785506233573, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00235, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 106683 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.23940840363502502, + "learning_rate": 6.521548694236384e-05, + "loss": 0.003448254195973277, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00345, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107121 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.26199451088905334, + "learning_rate": 6.493256429322259e-05, + "loss": 0.005110536701977253, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00512, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 107561 + }, + { + "epoch": 0.921875, + "grad_norm": 0.19866062700748444, + "learning_rate": 6.464922830953799e-05, + "loss": 0.004088824614882469, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0041, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108018 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.024193251505494118, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0006136820884421468, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00061, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 108446 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.05543803051114082, + "learning_rate": 6.408136351831592e-05, + "loss": 0.0012521358439698815, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00125, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.3, + "tokens/trainable": 108866 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.7693765163421631, + "learning_rate": 6.379685834195036e-05, + "loss": 0.010034173727035522, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01008, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 109316 + }, + { + "epoch": 0.9375, + "grad_norm": 0.06719055026769638, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014942979905754328, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0015, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 109787 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.021327359601855278, + "learning_rate": 6.32267616243259e-05, + "loss": 0.0003477554419077933, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110246 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.015047608874738216, + "learning_rate": 6.294119380711849e-05, + "loss": 0.00033737512421794236, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00034, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.27, + "tokens/trainable": 110667 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.14928486943244934, + "learning_rate": 6.265529552442209e-05, + "loss": 0.018782632425427437, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01896, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 111154 + }, + { + "epoch": 0.953125, + "grad_norm": 0.005887618288397789, + "learning_rate": 6.236907867363127e-05, + "loss": 0.00017539145483169705, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00018, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 111611 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.04487505182623863, + "learning_rate": 6.208255516539749e-05, + "loss": 0.0005144703900441527, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00051, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 112102 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.3277672231197357, + "learning_rate": 6.179573692313344e-05, + "loss": 0.010591475293040276, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.01065, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 112569 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.010867438279092312, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0003141904016956687, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00031, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 113023 + }, + { + "epoch": 0.96875, + "grad_norm": 0.011889747343957424, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0003558638272807002, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 113464 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.040197841823101044, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0003707111463882029, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00037, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 113908 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.01977125182747841, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000563526526093483, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00056, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 114348 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.04650595411658287, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.0004946058033965528, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00049, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 114830 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3321515619754791, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0047949617728590965, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00481, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.25, + "tokens/trainable": 115278 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.248526468873024, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.009970192797482014, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01002, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 115699 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.01950794644653797, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.000543278525583446, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00054, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.85, + "tokens/trainable": 116133 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.013262342661619186, + "learning_rate": 5.920308275189541e-05, + "loss": 0.00043126061791554093, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.89, + "tokens/trainable": 116591 + }, + { + "epoch": 1.0, + "grad_norm": 0.008060271851718426, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0002502937277313322, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00025, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117039 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.006059759296476841, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.000195491302292794, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 117495 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026264909654855728, + "learning_rate": 5.833528413179249e-05, + "loss": 0.00039901715354062617, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0004, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.58, + "tokens/trainable": 117925 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.035495247691869736, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0005823379615321755, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00058, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 118373 + }, + { + "epoch": 1.015625, + "grad_norm": 0.004000507295131683, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00013823891640640795, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 118816 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.03404498100280762, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0006110825343057513, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00061, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 119263 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.018318751826882362, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0002658100565895438, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00027, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 119772 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.21364416182041168, + "learning_rate": 5.688633799118971e-05, + "loss": 0.002921548904851079, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00293, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 120269 + }, + { + "epoch": 1.03125, + "grad_norm": 0.014080969616770744, + "learning_rate": 5.659626500889066e-05, + "loss": 0.000284558511339128, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00028, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 120746 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.018712079152464867, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0003485676134005189, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 121204 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.16164301335811615, + "learning_rate": 5.601593183686955e-05, + "loss": 0.004497133661061525, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00451, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 121670 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.006513051688671112, + "learning_rate": 5.572569579717961e-05, + "loss": 0.00010482803918421268, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0001, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.18, + "tokens/trainable": 122183 + }, + { + "epoch": 1.046875, + "grad_norm": 0.008712893351912498, + "learning_rate": 5.543542955832538e-05, + "loss": 0.0001421938359271735, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00014, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 122645 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.006345031317323446, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00020790408598259091, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00021, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 123105 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.003045592224225402, + "learning_rate": 5.485485480053015e-05, + "loss": 8.933185745263472e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 123595 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.10080692917108536, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0011391319567337632, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00114, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 124077 + }, + { + "epoch": 1.0625, + "grad_norm": 0.01787402294576168, + "learning_rate": 5.42743042028204e-05, + "loss": 0.000280556152574718, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00028, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.72, + "tokens/trainable": 124547 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.007194901816546917, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.00012255870387889445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00012, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 124977 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.05285244435071945, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0003489117370918393, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00035, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 125438 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.004284605849534273, + "learning_rate": 5.340373499110935e-05, + "loss": 0.000126057377201505, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00013, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.58, + "tokens/trainable": 125889 + }, + { + "epoch": 1.078125, + "grad_norm": 0.006148909218609333, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.00012465259351301938, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00012, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 126363 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.006904160603880882, + "learning_rate": 5.282366752473479e-05, + "loss": 0.00013310338545124978, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00013, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.11, + "tokens/trainable": 126828 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.001829590299166739, + "learning_rate": 5.2533763606737005e-05, + "loss": 4.333024116931483e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00004, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 127331 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.07432714849710464, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0006949692033231258, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0007, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 127798 + }, + { + "epoch": 1.09375, + "grad_norm": 0.048953574150800705, + "learning_rate": 5.195427572104522e-05, + "loss": 0.0004470825952012092, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00045, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 128247 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.0006554989377036691, + "learning_rate": 5.166471586820751e-05, + "loss": 1.6557103663217276e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 128729 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01792880706489086, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00023528770543634892, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00024, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 129203 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.07039739936590195, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0005169064970687032, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00052, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 129653 + }, + { + "epoch": 1.109375, + "grad_norm": 0.25704729557037354, + "learning_rate": 5.079691724810461e-05, + "loss": 0.001955911749973893, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00196, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 130100 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.0025027310475707054, + "learning_rate": 5.0507984812754684e-05, + "loss": 4.597695806296542e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.52, + "tokens/trainable": 130543 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.002928070956841111, + "learning_rate": 5.021923930849237e-05, + "loss": 6.182275683386251e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 131000 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.0009919465519487858, + "learning_rate": 4.99306927511967e-05, + "loss": 2.293481884407811e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00002, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.09, + "tokens/trainable": 131449 + }, + { + "epoch": 1.125, + "grad_norm": 0.0005833669565618038, + "learning_rate": 4.964235714846775e-05, + "loss": 1.711631557554938e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.12, + "tokens/trainable": 131923 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0012934672413393855, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.084562922595069e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 132397 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.0005865129642188549, + "learning_rate": 4.90663667927174e-05, + "loss": 1.5193030776572414e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.47, + "tokens/trainable": 132846 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.004269362892955542, + "learning_rate": 4.877873600900581e-05, + "loss": 5.181954838917591e-05, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 133347 + }, + { + "epoch": 1.140625, + "grad_norm": 0.0013308345805853605, + "learning_rate": 4.849136411748306e-05, + "loss": 2.814647086779587e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00003, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 133779 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.002314274897798896, + "learning_rate": 4.8204263076866574e-05, + "loss": 3.513288538670167e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 134251 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.0008702389895915985, + "learning_rate": 4.791744483460251e-05, + "loss": 2.252616104669869e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.62, + "tokens/trainable": 134683 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.018964840099215508, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0001443990768166259, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00014, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 135165 + }, + { + "epoch": 1.15625, + "grad_norm": 0.011429708451032639, + "learning_rate": 4.7344704475577916e-05, + "loss": 8.412318129558116e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00008, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 135599 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.12274184077978134, + "learning_rate": 4.705880619288153e-05, + "loss": 0.0006645542453043163, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 136038 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.003483664011582732, + "learning_rate": 4.677323837567412e-05, + "loss": 6.217554619070143e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.99, + "tokens/trainable": 136495 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.0016517788171768188, + "learning_rate": 4.6488012907598146e-05, + "loss": 3.058795482502319e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 136953 + }, + { + "epoch": 1.171875, + "grad_norm": 0.3304859697818756, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002893730765208602, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0029, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 137412 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.13385294377803802, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0007642972050234675, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00076, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.66, + "tokens/trainable": 137907 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0037263180129230022, + "learning_rate": 4.5634509217923135e-05, + "loss": 5.128432167111896e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 138383 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.006775988731533289, + "learning_rate": 4.535077169046201e-05, + "loss": 2.8198699510539882e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 138842 + }, + { + "epoch": 1.1875, + "grad_norm": 0.004235657397657633, + "learning_rate": 4.506743570677743e-05, + "loss": 4.3131229176651686e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 139269 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.0024381300900131464, + "learning_rate": 4.478451305763618e-05, + "loss": 4.280401481082663e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00004, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139736 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.0007152331527322531, + "learning_rate": 4.450201551660454e-05, + "loss": 1.5340305253630504e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.15, + "tokens/trainable": 140226 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.41952189803123474, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.005223053507506847, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00524, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 140664 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0014089028118178248, + "learning_rate": 4.393834276419352e-05, + "loss": 2.587199560366571e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 30.02, + "tokens/trainable": 141115 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.5370892882347107, + "learning_rate": 4.36571910095382e-05, + "loss": 0.002367620589211583, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00237, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 29.03, + "tokens/trainable": 141530 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.001014610636048019, + "learning_rate": 4.337651127546448e-05, + "loss": 1.8852246284950525e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 141964 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.011256729252636433, + "learning_rate": 4.3096315242201736e-05, + "loss": 6.678313366137445e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 142428 + }, + { + "epoch": 1.21875, + "grad_norm": 0.004441538825631142, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.578820127993822e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 142894 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.013733429834246635, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0001298388233408332, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.49, + "tokens/trainable": 143368 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.46937739849090576, + "learning_rate": 4.225874584473174e-05, + "loss": 0.008783280849456787, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00882, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 33.0, + "tokens/trainable": 143824 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0013688476756215096, + "learning_rate": 4.19806010071587e-05, + "loss": 2.1897705664741807e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00002, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 144306 + }, + { + "epoch": 1.234375, + "grad_norm": 0.0007963742245920002, + "learning_rate": 4.170299795992081e-05, + "loss": 1.3051090718363412e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00001, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 144763 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.002978365868330002, + "learning_rate": 4.142594825521398e-05, + "loss": 2.2917138267075643e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.61, + "tokens/trainable": 145255 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.0005564937018789351, + "learning_rate": 4.114946342220728e-05, + "loss": 1.2049408724124078e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 145714 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.4178679287433624, + "learning_rate": 4.087355496656321e-05, + "loss": 0.008204679936170578, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00824, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.21265174448490143, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0010491310385987163, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00105, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.4667704999446869, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.007591300178319216, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00762, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.45, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.40933945775032043, + "learning_rate": 4.004940255778431e-05, + "loss": 0.011366574093699455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01143, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.03840261325240135, + "learning_rate": 3.977591418134619e-05, + "loss": 0.00024993589613586664, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00922832265496254, + "learning_rate": 3.95030593412612e-05, + "loss": 6.446735642384738e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00006, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.19, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.007167867384850979, + "learning_rate": 3.923084939213296e-05, + "loss": 0.0001015613743220456, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.64, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006592086516320705, + "learning_rate": 3.895929566172861e-05, + "loss": 4.839149187318981e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.2736883759498596, + "learning_rate": 3.868840945050728e-05, + "loss": 0.006358759012073278, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00638, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.59, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.1449279934167862, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0009410807979293168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00094, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.009757250547409058, + "learning_rate": 3.814868464809027e-05, + "loss": 0.00017852694145403802, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.03571556136012077, + "learning_rate": 3.787986851704667e-05, + "loss": 0.0004097867349628359, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00041, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.22, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05014437809586525, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.0006581329507753253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.2, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.15289872884750366, + "learning_rate": 3.734438472750619e-05, + "loss": 0.0020632329396903515, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00207, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.15180747210979462, + "learning_rate": 3.707773935267552e-05, + "loss": 0.0012369159376248717, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00124, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.11855534464120865, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0021295640617609024, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00213, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.06385980546474457, + "learning_rate": 3.654669712344384e-05, + "loss": 0.0008570625213906169, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00086, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.15227577090263367, + "learning_rate": 3.628232236787763e-05, + "loss": 0.003443875815719366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00345, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.25938382744789124, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.002705808263272047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00271, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.08946357667446136, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0004575806378852576, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00046, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008916565217077732, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00013453952851705253, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00013, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.91, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.006945237051695585, + "learning_rate": 3.5232722063479914e-05, + "loss": 7.873260619817302e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.009922013618052006, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010111248411703855, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.02600066177546978, + "learning_rate": 3.471281389826491e-05, + "loss": 0.00031280232360586524, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00031, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.016229942440986633, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.00017198207206092775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00017, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.016324417665600777, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.00016402543406002223, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.32190635800361633, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.004134346265345812, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00414, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.00600312277674675, + "learning_rate": 3.3683214232914404e-05, + "loss": 8.641595195513219e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.13886022567749023, + "learning_rate": 3.342800532426873e-05, + "loss": 0.0013213125057518482, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00132, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.023262491449713707, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002245823125122115, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00022, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.78, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.208764910697937, + "learning_rate": 3.292029118616024e-05, + "loss": 0.001788674620911479, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00179, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.62, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.001404186594299972, + "learning_rate": 3.266780708475511e-05, + "loss": 2.7747140848077834e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.012655309401452541, + "learning_rate": 3.241625231705354e-05, + "loss": 0.0001932395389303565, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00019, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.63, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.030525848269462585, + "learning_rate": 3.216563735127618e-05, + "loss": 0.000371504167560488, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00037, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0018862849101424217, + "learning_rate": 3.191597261653475e-05, + "loss": 3.8734902773285285e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.0015944828046485782, + "learning_rate": 3.166726850239794e-05, + "loss": 1.4231935892894398e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00001, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.004869155120104551, + "learning_rate": 3.141953535845912e-05, + "loss": 7.406627264572307e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.01870773173868656, + "learning_rate": 3.11727834939056e-05, + "loss": 0.0001340095914201811, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00013, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.08653457462787628, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0008557374821975827, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00086, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.49505236744880676, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.007139122113585472, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00716, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.033166300505399704, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00024455596576444805, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00024, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.09, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.026151562109589577, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00018419645493850112, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00018, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.5249567627906799, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.03347098454833031, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03404, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.008504888974130154, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00012848488404415548, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00013, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.08, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.006400417070835829, + "learning_rate": 2.9473853553877484e-05, + "loss": 7.728980563115329e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.010285867378115654, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00018826585437636822, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.05493743345141411, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.0002043755230261013, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.00775935361161828, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00012459162098821253, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.21, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.05490761622786522, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.000744640186894685, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.3, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.006251805927604437, + "learning_rate": 2.829199644117484e-05, + "loss": 0.00010329978249501437, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0001, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.01030012033879757, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001805450883693993, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.007539911661297083, + "learning_rate": 2.782696506053033e-05, + "loss": 0.00016556130140088499, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00017, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.004288971424102783, + "learning_rate": 2.7596140715257824e-05, + "loss": 0.0001009388652164489, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.17, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.01721479743719101, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0003073053085245192, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00031, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.008515549823641777, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.00017193684470839798, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00017, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.009272439405322075, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00021900574211031199, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00022, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.034540578722953796, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0005546602769754827, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00055, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.013745410367846489, + "learning_rate": 2.645931522709877e-05, + "loss": 0.0003212787851225585, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00032, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.27, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.005436223931610584, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0001245749299414456, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00012, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.010458163917064667, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0002482631243765354, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00025, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.006481671240180731, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0001577583752805367, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.012174229137599468, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00027598816086538136, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00028, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.011468137614428997, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00023577807587571442, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00024, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.005027628969401121, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.00010462553473189473, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.005808121990412474, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00013498810585588217, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00013, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.01, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0358550138771534, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.0006072560790926218, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00061, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.09168751537799835, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.0007024870719760656, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0007, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.35, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.014628843404352665, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00020378318731673062, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.05326993018388748, + "learning_rate": 2.406442693028651e-05, + "loss": 0.00035949741140939295, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00036, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.008257771842181683, + "learning_rate": 2.3854255584458547e-05, + "loss": 9.612501162337139e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.03791302442550659, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.0004905672976747155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00049, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2622528076171875, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.003920107148587704, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00393, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.06083231046795845, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.0009291216847486794, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00093, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.014715000987052917, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00020523657440207899, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00021, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.01740424521267414, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00030176169821061194, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.0003, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.005717985797673464, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00011773040023399517, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00012, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.005951862782239914, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.0001332889514742419, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00013, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.0560171976685524, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00048283624346368015, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.013027072884142399, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00022208130394574255, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00022, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.0371403768658638, + "learning_rate": 2.182524383352446e-05, + "loss": 0.0005342152435332537, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00053, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.68, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.004872396122664213, + "learning_rate": 2.1629798596457056e-05, + "loss": 7.259240373969078e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00007, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.010584473609924316, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.00017825220129452646, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00018, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.33, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.007694529369473457, + "learning_rate": 2.124308220868431e-05, + "loss": 7.509582792408764e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00008, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.02000581845641136, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0002455090289004147, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00025, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.0213518887758255, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.0002898501115851104, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00029, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.6, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.004943408537656069, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0001215632728417404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00012, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.39, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.012947587296366692, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001342954346910119, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00013, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.05408314988017082, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00036081415601074696, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.65, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.037419769912958145, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00042080413550138474, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00042, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.035673510283231735, + "learning_rate": 1.993423839463052e-05, + "loss": 0.000293174380203709, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00029, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0039419797249138355, + "learning_rate": 1.975303621298445e-05, + "loss": 8.023489499464631e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00008, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.05340631678700447, + "learning_rate": 1.957330080137385e-05, + "loss": 0.00029110765899531543, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.1, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.016526181250810623, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00015069378423504531, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.41, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.008226566947996616, + "learning_rate": 1.9218260145006073e-05, + "loss": 9.167128155240789e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00009, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.005769859533756971, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00011197607091162354, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00011, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.08041926473379135, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.000366955588106066, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00037, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.03303743153810501, + "learning_rate": 1.869688492349885e-05, + "loss": 0.0001731862430460751, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.008530828170478344, + "learning_rate": 1.85261050441233e-05, + "loss": 0.00011680425814120099, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00012, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 190216 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.0015226984396576881, + "learning_rate": 1.8356842992397304e-05, + "loss": 3.2423220545751974e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00003, + "step": 417, + "tokens/total": 12617168, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 190708 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.03927593678236008, + "learning_rate": 1.8189105812005714e-05, + "loss": 0.00023564108414575458, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 418, + "tokens/total": 12647680, + "tokens/train_per_sec_per_gpu": 31.01, + "tokens/trainable": 191126 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.06375617533922195, + "learning_rate": 1.802290048317732e-05, + "loss": 0.0004688606131821871, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00047, + "step": 419, + "tokens/total": 12677952, + "tokens/train_per_sec_per_gpu": 30.44, + "tokens/trainable": 191540 + }, + { + "epoch": 1.640625, + "grad_norm": 0.01704459637403488, + "learning_rate": 1.785823392239424e-05, + "loss": 0.00010123683023266494, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 420, + "tokens/total": 12708416, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.0012392516946420074, + "learning_rate": 1.7695112982104225e-05, + "loss": 2.009748641285114e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 421, + "tokens/total": 12738640, + "tokens/train_per_sec_per_gpu": 31.36, + "tokens/trainable": 192463 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.06435113400220871, + "learning_rate": 1.7533544450435433e-05, + "loss": 0.0006201984360814095, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00062, + "step": 422, + "tokens/total": 12769232, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 192918 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.04247846081852913, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.0001703215966699645, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00017, + "step": 423, + "tokens/total": 12799648, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 193425 + }, + { + "epoch": 1.65625, + "grad_norm": 0.0037633716128766537, + "learning_rate": 1.721509144218405e-05, + "loss": 7.026562525425106e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00007, + "step": 424, + "tokens/total": 12829920, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 193898 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.24774393439292908, + "learning_rate": 1.705822021773101e-05, + "loss": 0.0018032332882285118, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0018, + "step": 425, + "tokens/total": 12860112, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 194315 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.0021040684077888727, + "learning_rate": 1.69029279056068e-05, + "loss": 3.659192589111626e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00004, + "step": 426, + "tokens/total": 12890320, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 194796 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.001372064813040197, + "learning_rate": 1.6749220968158415e-05, + "loss": 2.2382890165317804e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00002, + "step": 427, + "tokens/total": 12920656, + "tokens/train_per_sec_per_gpu": 34.08, + "tokens/trainable": 195262 + }, + { + "epoch": 1.671875, + "grad_norm": 0.004309098701924086, + "learning_rate": 1.659710580175893e-05, + "loss": 5.155909457243979e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 428, + "tokens/total": 12951040, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 195680 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.03517591208219528, + "learning_rate": 1.644658873654133e-05, + "loss": 0.00021132534311618656, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00021, + "step": 429, + "tokens/total": 12981232, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 196145 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.005870590452104807, + "learning_rate": 1.629767603613508e-05, + "loss": 6.348404713207856e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00006, + "step": 430, + "tokens/total": 13011616, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 196626 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.2625119984149933, + "learning_rate": 1.615037389740547e-05, + "loss": 0.011768318712711334, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01184, + "step": 431, + "tokens/total": 13042208, + "tokens/train_per_sec_per_gpu": 31.22, + "tokens/trainable": 197067 + }, + { + "epoch": 1.6875, + "grad_norm": 0.007063967175781727, + "learning_rate": 1.600468845019576e-05, + "loss": 4.784313205163926e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00005, + "step": 432, + "tokens/total": 13072800, + "tokens/train_per_sec_per_gpu": 38.79, + "tokens/trainable": 197565 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.006292980629950762, + "learning_rate": 1.5860625757072092e-05, + "loss": 5.077052628621459e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00005, + "step": 433, + "tokens/total": 13103328, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 198015 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.0020431573502719402, + "learning_rate": 1.571819181307116e-05, + "loss": 3.991067933384329e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00004, + "step": 434, + "tokens/total": 13133472, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 198442 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.004255881067365408, + "learning_rate": 1.557739254545075e-05, + "loss": 5.153669189894572e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00005, + "step": 435, + "tokens/total": 13164064, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 198914 + }, + { + "epoch": 1.703125, + "grad_norm": 0.46879518032073975, + "learning_rate": 1.543823381344311e-05, + "loss": 0.013075481168925762, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.01316, + "step": 436, + "tokens/total": 13194464, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 199355 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.22727783024311066, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.0031720949336886406, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00318, + "step": 437, + "tokens/total": 13224992, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 199830 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.003598392242565751, + "learning_rate": 1.5164861051607254e-05, + "loss": 5.355676694307476e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00005, + "step": 438, + "tokens/total": 13255296, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 200338 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.04593397676944733, + "learning_rate": 1.5030658397935521e-05, + "loss": 0.00022313687077257782, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00022, + "step": 439, + "tokens/total": 13285344, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 200785 + }, + { + "epoch": 1.71875, + "grad_norm": 0.0011655986309051514, + "learning_rate": 1.4898119031716104e-05, + "loss": 2.53522812272422e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 440, + "tokens/total": 13315632, + "tokens/train_per_sec_per_gpu": 36.4, + "tokens/trainable": 201267 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.010816111229360104, + "learning_rate": 1.476724846845306e-05, + "loss": 0.0002002313849516213, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0002, + "step": 441, + "tokens/total": 13346048, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 201734 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.008938372135162354, + "learning_rate": 1.463805215420471e-05, + "loss": 0.00011111146159237251, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 442, + "tokens/total": 13376304, + "tokens/train_per_sec_per_gpu": 30.45, + "tokens/trainable": 202174 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.010652483440935612, + "learning_rate": 1.451053546535705e-05, + "loss": 0.000100844117696397, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 443, + "tokens/total": 13406720, + "tokens/train_per_sec_per_gpu": 36.34, + "tokens/trainable": 202612 + }, + { + "epoch": 1.734375, + "grad_norm": 0.005689944606274366, + "learning_rate": 1.438470370840001e-05, + "loss": 0.00011003226973116398, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00011, + "step": 444, + "tokens/total": 13436464, + "tokens/train_per_sec_per_gpu": 31.09, + "tokens/trainable": 203020 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.24016952514648438, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.002352698938921094, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00236, + "step": 445, + "tokens/total": 13466672, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 203479 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.008882527239620686, + "learning_rate": 1.413811586531508e-05, + "loss": 0.0001598334638401866, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00016, + "step": 446, + "tokens/total": 13496736, + "tokens/train_per_sec_per_gpu": 35.54, + "tokens/trainable": 203946 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.005755060818046331, + "learning_rate": 1.4017370040713884e-05, + "loss": 7.665179145988077e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 447, + "tokens/total": 13526928, + "tokens/train_per_sec_per_gpu": 37.64, + "tokens/trainable": 204382 + }, + { + "epoch": 1.75, + "grad_norm": 0.003990499302744865, + "learning_rate": 1.3898329670629645e-05, + "loss": 8.206715574488044e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00008, + "step": 448, + "tokens/total": 13557392, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 204830 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.0017717559821903706, + "learning_rate": 1.3780999708818058e-05, + "loss": 3.886378544848412e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00004, + "step": 449, + "tokens/total": 13587776, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 205300 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.004407316446304321, + "learning_rate": 1.3665385037857758e-05, + "loss": 6.286771531449631e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 450, + "tokens/total": 13618112, + "tokens/train_per_sec_per_gpu": 30.74, + "tokens/trainable": 205726 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.004234767984598875, + "learning_rate": 1.3551490468947126e-05, + "loss": 0.00010631309851305559, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00011, + "step": 451, + "tokens/total": 13648512, + "tokens/train_per_sec_per_gpu": 31.77, + "tokens/trainable": 206163 + }, + { + "epoch": 1.765625, + "grad_norm": 0.19428566098213196, + "learning_rate": 1.3439320741704075e-05, + "loss": 0.008113588206470013, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00815, + "step": 452, + "tokens/total": 13678704, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 206619 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.35631781816482544, + "learning_rate": 1.3328880523968808e-05, + "loss": 0.004253438673913479, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00426, + "step": 453, + "tokens/total": 13709232, + "tokens/train_per_sec_per_gpu": 37.36, + "tokens/trainable": 207101 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.004516016226261854, + "learning_rate": 1.3220174411609587e-05, + "loss": 5.911413245485164e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 454, + "tokens/total": 13739600, + "tokens/train_per_sec_per_gpu": 35.83, + "tokens/trainable": 207546 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.007509376388043165, + "learning_rate": 1.3113206928331471e-05, + "loss": 0.00010886647214647382, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00011, + "step": 455, + "tokens/total": 13769936, + "tokens/train_per_sec_per_gpu": 34.43, + "tokens/trainable": 207989 + }, + { + "epoch": 1.78125, + "grad_norm": 0.0016118305502459407, + "learning_rate": 1.300798252548806e-05, + "loss": 3.1790965294931084e-05, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00003, + "step": 456, + "tokens/total": 13800240, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 208402 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.0032239535357803106, + "learning_rate": 1.2904505581896265e-05, + "loss": 4.2626015783753246e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00004, + "step": 457, + "tokens/total": 13830512, + "tokens/train_per_sec_per_gpu": 36.74, + "tokens/trainable": 208904 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.0048893350176513195, + "learning_rate": 1.2802780403654082e-05, + "loss": 6.955669960007071e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 458, + "tokens/total": 13860832, + "tokens/train_per_sec_per_gpu": 33.78, + "tokens/trainable": 209382 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.003503630170598626, + "learning_rate": 1.2702811223961408e-05, + "loss": 7.862070197006688e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00008, + "step": 459, + "tokens/total": 13890992, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 209833 + }, + { + "epoch": 1.796875, + "grad_norm": 0.34274882078170776, + "learning_rate": 1.2604602202943861e-05, + "loss": 0.010280012153089046, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01033, + "step": 460, + "tokens/total": 13921424, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 210284 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.15473094582557678, + "learning_rate": 1.2508157427479686e-05, + "loss": 0.0024835069198161364, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00249, + "step": 461, + "tokens/total": 13951504, + "tokens/train_per_sec_per_gpu": 34.19, + "tokens/trainable": 210769 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.0028502352070063353, + "learning_rate": 1.2413480911029655e-05, + "loss": 5.3788880904903635e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00005, + "step": 462, + "tokens/total": 13979728, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 211193 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.0012627877295017242, + "learning_rate": 1.2320576593470082e-05, + "loss": 3.094038038398139e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00003, + "step": 463, + "tokens/total": 14009936, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 211655 + }, + { + "epoch": 1.8125, + "grad_norm": 0.0036251619458198547, + "learning_rate": 1.2229448340928828e-05, + "loss": 8.078858081717044e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00008, + "step": 464, + "tokens/total": 14039872, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 212085 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.017260659486055374, + "learning_rate": 1.2140099945624458e-05, + "loss": 0.0002048999012913555, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0002, + "step": 465, + "tokens/total": 14070096, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 212570 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.2109188288450241, + "learning_rate": 1.205253512570841e-05, + "loss": 0.006965605076402426, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00699, + "step": 466, + "tokens/total": 14100192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 213011 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.052500564604997635, + "learning_rate": 1.1966757525110255e-05, + "loss": 0.0004830099060200155, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00048, + "step": 467, + "tokens/total": 14130448, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 213450 + }, + { + "epoch": 1.828125, + "grad_norm": 0.005511680152267218, + "learning_rate": 1.1882770713386095e-05, + "loss": 8.27932235551998e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00008, + "step": 468, + "tokens/total": 14160480, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 213898 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.14864490926265717, + "learning_rate": 1.180057818556998e-05, + "loss": 0.0016087992116808891, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00161, + "step": 469, + "tokens/total": 14191152, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 214366 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.004650567192584276, + "learning_rate": 1.1720183362028494e-05, + "loss": 9.897982818074524e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 470, + "tokens/total": 14219392, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 214800 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.06575815379619598, + "learning_rate": 1.1641589588318387e-05, + "loss": 0.0007422211347147822, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00074, + "step": 471, + "tokens/total": 14249920, + "tokens/train_per_sec_per_gpu": 35.59, + "tokens/trainable": 215292 + }, + { + "epoch": 1.84375, + "grad_norm": 0.002274462953209877, + "learning_rate": 1.1564800135047418e-05, + "loss": 5.863649130333215e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00006, + "step": 472, + "tokens/total": 14280048, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 215741 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.002968342276290059, + "learning_rate": 1.148981819773816e-05, + "loss": 6.629896233789623e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00007, + "step": 473, + "tokens/total": 14310432, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 216194 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.004850646015256643, + "learning_rate": 1.1416646896695086e-05, + "loss": 9.518570732325315e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0001, + "step": 474, + "tokens/total": 14340496, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 216634 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.023391559720039368, + "learning_rate": 1.1345289276874717e-05, + "loss": 0.00047331867972388864, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 475, + "tokens/total": 14370720, + "tokens/train_per_sec_per_gpu": 30.6, + "tokens/trainable": 217040 + }, + { + "epoch": 1.859375, + "grad_norm": 0.00271675456315279, + "learning_rate": 1.1275748307758873e-05, + "loss": 5.663794945576228e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 476, + "tokens/total": 14401136, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 217530 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.018746392801404, + "learning_rate": 1.1208026883231147e-05, + "loss": 0.00019438326125964522, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00019, + "step": 477, + "tokens/total": 14431360, + "tokens/train_per_sec_per_gpu": 34.65, + "tokens/trainable": 217988 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.12623320519924164, + "learning_rate": 1.1142127821456433e-05, + "loss": 0.0016139973886311054, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00162, + "step": 478, + "tokens/total": 14461968, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 218461 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.00717399176210165, + "learning_rate": 1.1078053864763674e-05, + "loss": 0.00011526358139235526, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00012, + "step": 479, + "tokens/total": 14491984, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 218909 + }, + { + "epoch": 1.875, + "grad_norm": 0.05601891130208969, + "learning_rate": 1.1015807679531756e-05, + "loss": 0.0003784724394790828, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00038, + "step": 480, + "tokens/total": 14522544, + "tokens/train_per_sec_per_gpu": 34.19, + "tokens/trainable": 219350 + }, + { + "epoch": 1.87890625, + "grad_norm": 0.0026196292601525784, + "learning_rate": 1.0955391856078528e-05, + "loss": 6.231790757738054e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00006, + "step": 481, + "tokens/total": 14552560, + "tokens/train_per_sec_per_gpu": 29.63, + "tokens/trainable": 219776 + }, + { + "epoch": 1.8828125, + "grad_norm": 0.034271057695150375, + "learning_rate": 1.0896808908553007e-05, + "loss": 0.00036835690843872726, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00037, + "step": 482, + "tokens/total": 14583056, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 220284 + }, + { + "epoch": 1.88671875, + "grad_norm": 0.38923949003219604, + "learning_rate": 1.0840061274830763e-05, + "loss": 0.002996535040438175, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.003, + "step": 483, + "tokens/total": 14613152, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 220714 + }, + { + "epoch": 1.890625, + "grad_norm": 0.008355424739420414, + "learning_rate": 1.0785151316412473e-05, + "loss": 0.00015630274720024318, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 484, + "tokens/total": 14643328, + "tokens/train_per_sec_per_gpu": 35.08, + "tokens/trainable": 221156 + }, + { + "epoch": 1.89453125, + "grad_norm": 0.0020718539599329233, + "learning_rate": 1.0732081318325639e-05, + "loss": 4.889669071417302e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00005, + "step": 485, + "tokens/total": 14673488, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 221605 + }, + { + "epoch": 1.8984375, + "grad_norm": 0.005701969377696514, + "learning_rate": 1.0680853489029501e-05, + "loss": 6.45708350930363e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00006, + "step": 486, + "tokens/total": 14704000, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 222042 + }, + { + "epoch": 1.90234375, + "grad_norm": 0.0030111433006823063, + "learning_rate": 1.0631469960323152e-05, + "loss": 7.045763777568936e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00007, + "step": 487, + "tokens/total": 14733984, + "tokens/train_per_sec_per_gpu": 32.15, + "tokens/trainable": 222474 + }, + { + "epoch": 1.90625, + "grad_norm": 0.0025332679506391287, + "learning_rate": 1.0583932787256783e-05, + "loss": 5.044174031354487e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00005, + "step": 488, + "tokens/total": 14764400, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 222967 + }, + { + "epoch": 1.91015625, + "grad_norm": 0.01333449874073267, + "learning_rate": 1.0538243948046206e-05, + "loss": 7.783513137837872e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00008, + "step": 489, + "tokens/total": 14794672, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 223420 + }, + { + "epoch": 1.9140625, + "grad_norm": 0.019479215145111084, + "learning_rate": 1.0494405343990523e-05, + "loss": 0.0001881857169792056, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00019, + "step": 490, + "tokens/total": 14824784, + "tokens/train_per_sec_per_gpu": 29.43, + "tokens/trainable": 223864 + }, + { + "epoch": 1.91796875, + "grad_norm": 0.0038831171113997698, + "learning_rate": 1.0452418799392985e-05, + "loss": 7.516246841987595e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00008, + "step": 491, + "tokens/total": 14855056, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 224321 + }, + { + "epoch": 1.921875, + "grad_norm": 0.015537051483988762, + "learning_rate": 1.0412286061485102e-05, + "loss": 7.350469240918756e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00007, + "step": 492, + "tokens/total": 14885392, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 224802 + }, + { + "epoch": 1.92578125, + "grad_norm": 0.0036717222537845373, + "learning_rate": 1.03740088003539e-05, + "loss": 8.332736615557224e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 493, + "tokens/total": 14915760, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 225284 + }, + { + "epoch": 1.9296875, + "grad_norm": 0.09004666656255722, + "learning_rate": 1.0337588608872463e-05, + "loss": 0.001120982225984335, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00112, + "step": 494, + "tokens/total": 14946192, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 225733 + }, + { + "epoch": 1.93359375, + "grad_norm": 0.0008198385476134717, + "learning_rate": 1.0303027002633622e-05, + "loss": 2.1550637029577047e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00002, + "step": 495, + "tokens/total": 14976272, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 226172 + }, + { + "epoch": 1.9375, + "grad_norm": 0.004506978671997786, + "learning_rate": 1.0270325419886884e-05, + "loss": 5.0953691243194044e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 496, + "tokens/total": 15006544, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 226636 + }, + { + "epoch": 1.94140625, + "grad_norm": 0.09322372078895569, + "learning_rate": 1.0239485221478599e-05, + "loss": 0.000592258817050606, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00059, + "step": 497, + "tokens/total": 15037040, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 227067 + }, + { + "epoch": 1.9453125, + "grad_norm": 0.5627581477165222, + "learning_rate": 1.0210507690795292e-05, + "loss": 0.008054064586758614, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00809, + "step": 498, + "tokens/total": 15067344, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 227532 + }, + { + "epoch": 1.94921875, + "grad_norm": 0.0026798928156495094, + "learning_rate": 1.0183394033710305e-05, + "loss": 5.531937858904712e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00006, + "step": 499, + "tokens/total": 15097648, + "tokens/train_per_sec_per_gpu": 36.75, + "tokens/trainable": 228007 + }, + { + "epoch": 1.953125, + "grad_norm": 0.04757218807935715, + "learning_rate": 1.0158145378533583e-05, + "loss": 0.00010594214836601168, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 500, + "tokens/total": 15128000, + "tokens/train_per_sec_per_gpu": 30.98, + "tokens/trainable": 228457 + }, + { + "epoch": 1.95703125, + "grad_norm": 0.017352212220430374, + "learning_rate": 1.0134762775964726e-05, + "loss": 0.0002479254035279155, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00025, + "step": 501, + "tokens/total": 15158352, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 228928 + }, + { + "epoch": 1.9609375, + "grad_norm": 0.003859007963910699, + "learning_rate": 1.0113247199049278e-05, + "loss": 5.532489740289748e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.00006, + "step": 502, + "tokens/total": 15189136, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 229368 + }, + { + "epoch": 1.96484375, + "grad_norm": 0.007725914474576712, + "learning_rate": 1.0093599543138205e-05, + "loss": 0.00010436305456096306, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0001, + "step": 503, + "tokens/total": 15219616, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 229815 + }, + { + "epoch": 1.96875, + "grad_norm": 0.0022384016774594784, + "learning_rate": 1.0075820625850675e-05, + "loss": 5.345371755538508e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00005, + "step": 504, + "tokens/total": 15250032, + "tokens/train_per_sec_per_gpu": 38.98, + "tokens/trainable": 230319 + }, + { + "epoch": 1.97265625, + "grad_norm": 0.003804351668804884, + "learning_rate": 1.0059911187040013e-05, + "loss": 6.216375186340883e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00006, + "step": 505, + "tokens/total": 15280592, + "tokens/train_per_sec_per_gpu": 37.64, + "tokens/trainable": 230807 + }, + { + "epoch": 1.9765625, + "grad_norm": 0.002130430657416582, + "learning_rate": 1.0045871888762893e-05, + "loss": 3.884470061166212e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 506, + "tokens/total": 15310816, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 231257 + }, + { + "epoch": 1.98046875, + "grad_norm": 0.006141587160527706, + "learning_rate": 1.003370331525184e-05, + "loss": 8.762093784753233e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00009, + "step": 507, + "tokens/total": 15341424, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 231746 + }, + { + "epoch": 1.984375, + "grad_norm": 0.0012427824549376965, + "learning_rate": 1.002340597289085e-05, + "loss": 3.193254815414548e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.33, + "memory/max_allocated (GiB)": 33.33, + "ppl": 1.00003, + "step": 508, + "tokens/total": 15369664, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 232172 + }, + { + "epoch": 1.98828125, + "grad_norm": 0.0014760087942704558, + "learning_rate": 1.0014980290194387e-05, + "loss": 3.683042450575158e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00004, + "step": 509, + "tokens/total": 15400496, + "tokens/train_per_sec_per_gpu": 37.21, + "tokens/trainable": 232680 + }, + { + "epoch": 1.9921875, + "grad_norm": 0.14180026948451996, + "learning_rate": 1.0008426617789489e-05, + "loss": 0.0011464458657428622, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00115, + "step": 510, + "tokens/total": 15430864, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 233169 + }, + { + "epoch": 1.99609375, + "grad_norm": 0.007576049771159887, + "learning_rate": 1.0003745228401215e-05, + "loss": 9.165602386929095e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 511, + "tokens/total": 15461232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 233609 + }, + { + "epoch": 2.0, + "grad_norm": 0.0015405165031552315, + "learning_rate": 1.0000936316841296e-05, + "loss": 3.871887020068243e-05, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00004, + "step": 512, + "tokens/total": 15491760, + "tokens/train_per_sec_per_gpu": 36.36, + "tokens/trainable": 234078 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 1.0515162810795494e+18, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..311c3f1be1a00ddc63f94c3e7c00ea79013fbdda --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d092c1c8bcaa54b4594cf8f8ead3539e8d51330ff99d4905723d4775cb79b21c +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..24c59558c9bf28fff2fdf28d1054bf3116b721fe --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3fa61d10fd5ac90b3f3b4777fc9b7f4aecd554eb16b5eee6166b47cf2e7b7697 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..29fe8c0957f6a29bef030536a0d836c14dd13199 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21d00dc563d713980dd7a1d4b126d93ab060fe78b7ba0c4c92f92cd25a992569 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..34b7fc400f1006f136b1c89d6c58cf3d17830d72 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c053d5c6e293421f32ec74be7fd526b45c9c8622 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokens_state.json @@ -0,0 +1 @@ +{"total": 1940848, "trainable": 29183} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..fb3df226fa384f186a3d005aed6d44d47294156d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/trainer_state.json @@ -0,0 +1,930 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.25, + "eval_steps": 500, + "global_step": 64, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.3173669557885491e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/README.md b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/adapter_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..dd0739b2a8e28ba0bdd94a8d6ece5f5b2c2c1c80 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "v_proj", + "down_proj", + "gate_proj", + "q_proj", + "o_proj", + "k_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a561521cd9d95f155b985326d4b9a915c4e30e58 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:26f7a02adece4e7a6b0de804d204e7d7a639684da7aa02367447a28508222208 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/chat_template.jinja b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/optimizer.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6acc72b782537caff0c2ce7814465458e2d2f7a8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a41bbcf0a8ae581c1cc569f7ca92ab83f2998f432cd80d97b7b58ec63d949fd8 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/rng_state.pth b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..c033f4347b764091af577cf5043dff72a382d3f5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:58b2d07d478cf0253d5619ad2e9d23f5e40a65de8495503676e2505cd9d2856a +size 14645 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/scheduler.pt b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..433f82b14836de77d366e2c961bc5bdacdc9eb63 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer_config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokens_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a8c8f671c130f9a8c18d776fdd5f367198489d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokens_state.json @@ -0,0 +1 @@ +{"total": 2906288, "trainable": 43824} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/trainer_state.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..5f802224877638120c7b4bd152e0c87f862ebbcb --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/trainer_state.json @@ -0,0 +1,1378 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.375, + "eval_steps": 500, + "global_step": 96, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 0.9943057298660278, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.43, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8786405920982361, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 0.8411452770233154, + "learning_rate": 8.000000000000001e-06, + "loss": 0.1128283143043518, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11944, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 0.8970848917961121, + "learning_rate": 1.2e-05, + "loss": 0.10135132074356079, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10667, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 0.7805498242378235, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11149105429649353, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.11794, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 0.9325005412101746, + "learning_rate": 2e-05, + "loss": 0.09598484635353088, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10074, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.02, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 0.9768337607383728, + "learning_rate": 2.4e-05, + "loss": 0.06007109954953194, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06191, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.0281821489334106, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.07311521470546722, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07585, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.29, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.3234435319900513, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.044495582580566406, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0455, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.2, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.8732832670211792, + "learning_rate": 3.6e-05, + "loss": 0.032459765672683716, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03299, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 1.0495871305465698, + "learning_rate": 4e-05, + "loss": 0.02137758582830429, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02161, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 2.064997673034668, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.04079515486955643, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.04164, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.9026453495025635, + "learning_rate": 4.8e-05, + "loss": 0.06431213766336441, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06643, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.4172568321228027, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.07758771628141403, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.08068, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6533877849578857, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.0820830836892128, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.08555, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.7390506267547607, + "learning_rate": 6e-05, + "loss": 0.053628768771886826, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.05509, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.12, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 0.8172504305839539, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016706150025129318, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01685, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 0.7143669724464417, + "learning_rate": 6.800000000000001e-05, + "loss": 0.021398911252617836, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02163, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 2.224132537841797, + "learning_rate": 7.2e-05, + "loss": 0.016565855592489243, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0167, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.01, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.9165221452713013, + "learning_rate": 7.6e-05, + "loss": 0.038831885904073715, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0396, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.89, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.9385339617729187, + "learning_rate": 8e-05, + "loss": 0.02326519787311554, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02354, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 1.3244878053665161, + "learning_rate": 8.4e-05, + "loss": 0.04611007124185562, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.04719, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.130301594734192, + "learning_rate": 8.800000000000001e-05, + "loss": 0.03128095716238022, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.03178, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.9159126281738281, + "learning_rate": 9.200000000000001e-05, + "loss": 0.042277175933122635, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.04318, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.63, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 0.5571489334106445, + "learning_rate": 9.6e-05, + "loss": 0.025266434997320175, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02559, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.08, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 0.8746730089187622, + "learning_rate": 0.0001, + "loss": 0.022446369752287865, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0227, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 1.505344271659851, + "learning_rate": 9.99990636831587e-05, + "loss": 0.02043886110186577, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02065, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 1.0260790586471558, + "learning_rate": 9.999625477159879e-05, + "loss": 0.021617872640490532, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02185, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 0.8115366101264954, + "learning_rate": 9.999157338221051e-05, + "loss": 0.020092125982046127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0203, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.5937158465385437, + "learning_rate": 9.998501970980562e-05, + "loss": 0.014808757230639458, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01492, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.7298683524131775, + "learning_rate": 9.997659402710915e-05, + "loss": 0.019437074661254883, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01963, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.17546068131923676, + "learning_rate": 9.996629668474818e-05, + "loss": 0.003190835705026984, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0032, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 0.8119296431541443, + "learning_rate": 9.995412811123711e-05, + "loss": 0.021450746804475784, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.02168, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.42, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.7597190141677856, + "learning_rate": 9.994008881295999e-05, + "loss": 0.02241288125514984, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02267, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 2.3171119689941406, + "learning_rate": 9.992417937414932e-05, + "loss": 0.021416299045085907, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.79, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.4047625958919525, + "learning_rate": 9.99064004568618e-05, + "loss": 0.005595909897238016, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00561, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.5865066051483154, + "learning_rate": 9.988675280095074e-05, + "loss": 0.010679518803954124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01074, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.583072304725647, + "learning_rate": 9.986523722403528e-05, + "loss": 0.02127654477953911, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0215, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.02, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.30342698097229, + "learning_rate": 9.984185462146642e-05, + "loss": 0.025776617228984833, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02611, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.1053258180618286, + "learning_rate": 9.98166059662897e-05, + "loss": 0.026606226339936256, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02696, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.6078025102615356, + "learning_rate": 9.978949230920472e-05, + "loss": 0.017936185002326965, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.0181, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.8376394510269165, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02379559725522995, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02408, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.4673662781715393, + "learning_rate": 9.972967458011312e-05, + "loss": 0.012735492549836636, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01282, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.35, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.38821765780448914, + "learning_rate": 9.96969729973664e-05, + "loss": 0.008668653666973114, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00871, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.7656640410423279, + "learning_rate": 9.966241139112754e-05, + "loss": 0.020857544615864754, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02108, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.55, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 0.6551951766014099, + "learning_rate": 9.96259911996461e-05, + "loss": 0.020156342536211014, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02036, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.35, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 1.2237868309020996, + "learning_rate": 9.958771393851491e-05, + "loss": 0.015184836462140083, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0153, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.571843147277832, + "learning_rate": 9.954758120060702e-05, + "loss": 0.01640632003545761, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01654, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.15, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 1.3066037893295288, + "learning_rate": 9.950559465600948e-05, + "loss": 0.017745690420269966, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0179, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.54, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.6320858597755432, + "learning_rate": 9.946175605195379e-05, + "loss": 0.009900711476802826, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00995, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.3268744647502899, + "learning_rate": 9.941606721274322e-05, + "loss": 0.006143633276224136, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00616, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 0.3406257629394531, + "learning_rate": 9.936853003967685e-05, + "loss": 0.004097095225006342, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00411, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 1.1476560831069946, + "learning_rate": 9.93191465109705e-05, + "loss": 0.028221281245350838, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02862, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.12, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.451351135969162, + "learning_rate": 9.926791868167438e-05, + "loss": 0.0078392019495368, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.00787, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.82, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8821313977241516, + "learning_rate": 9.921484868358753e-05, + "loss": 0.017667783424258232, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01782, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.6063156723976135, + "learning_rate": 9.915993872516924e-05, + "loss": 0.012062735855579376, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01214, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7402717471122742, + "learning_rate": 9.9103191091447e-05, + "loss": 0.030087217688560486, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03054, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.5932730436325073, + "learning_rate": 9.904460814392147e-05, + "loss": 0.013998140580952168, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0141, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 33.94, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.6833471059799194, + "learning_rate": 9.898419232046825e-05, + "loss": 0.009387579746544361, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00943, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.49946627020835876, + "learning_rate": 9.892194613523633e-05, + "loss": 0.014185642823576927, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01429, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.72, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.17088162899017334, + "learning_rate": 9.885787217854357e-05, + "loss": 0.003224549815058708, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00323, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.5056981444358826, + "learning_rate": 9.879197311676887e-05, + "loss": 0.012983381748199463, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01307, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.36, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 2.0115866661071777, + "learning_rate": 9.872425169224113e-05, + "loss": 0.02842879481613636, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02884, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.8811206221580505, + "learning_rate": 9.865471072312528e-05, + "loss": 0.016607588157057762, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01675, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.12346348166465759, + "learning_rate": 9.858335310330492e-05, + "loss": 0.003063689451664686, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00307, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.28303098678588867, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00448613939806819, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0045, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.7989771962165833, + "learning_rate": 9.843519986495259e-05, + "loss": 0.008180832490324974, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00821, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.66, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.12454555928707123, + "learning_rate": 9.835841041168162e-05, + "loss": 0.0016878236783668399, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00169, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.7390296459197998, + "learning_rate": 9.82798166379715e-05, + "loss": 0.016503486782312393, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01664, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.968937337398529, + "learning_rate": 9.819942181443002e-05, + "loss": 0.01685403287410736, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.017, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.3503779470920563, + "learning_rate": 9.811722928661392e-05, + "loss": 0.005242045037448406, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00526, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.5608110427856445, + "learning_rate": 9.803324247488975e-05, + "loss": 0.013865578919649124, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01396, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 1.2297694683074951, + "learning_rate": 9.794746487429161e-05, + "loss": 0.017295369878411293, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01745, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.72, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.1584348827600479, + "learning_rate": 9.785990005437554e-05, + "loss": 0.004732954781502485, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00474, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.44, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3246576488018036, + "learning_rate": 9.777055165907117e-05, + "loss": 0.013291046023368835, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01338, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 1.1803282499313354, + "learning_rate": 9.767942340652993e-05, + "loss": 0.014760692603886127, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01487, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 2.001532793045044, + "learning_rate": 9.758651908897035e-05, + "loss": 0.06761633604764938, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.06995, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.29207468032836914, + "learning_rate": 9.749184257252033e-05, + "loss": 0.006066338159143925, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00608, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.5561455488204956, + "learning_rate": 9.739539779705614e-05, + "loss": 0.033555109053850174, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03412, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.4088423550128937, + "learning_rate": 9.729718877603861e-05, + "loss": 0.014240232296288013, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01434, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.2712189853191376, + "learning_rate": 9.719721959634592e-05, + "loss": 0.011144062504172325, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01121, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.40229442715644836, + "learning_rate": 9.709549441810375e-05, + "loss": 0.012572492472827435, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01265, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.48, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 3.2695465087890625, + "learning_rate": 9.699201747451195e-05, + "loss": 0.012744108214974403, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01283, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.42963695526123047, + "learning_rate": 9.688679307166854e-05, + "loss": 0.014645008370280266, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01475, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.6282569169998169, + "learning_rate": 9.677982558839042e-05, + "loss": 0.009028157219290733, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00907, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.5804964900016785, + "learning_rate": 9.66711194760312e-05, + "loss": 0.010930215008556843, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01099, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.71, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.3565916121006012, + "learning_rate": 9.656067925829593e-05, + "loss": 0.011934653855860233, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01201, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.5799129009246826, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010369446128606796, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01042, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.76, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.4171448051929474, + "learning_rate": 9.633461496214225e-05, + "loss": 0.012263461016118526, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01234, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.11327923834323883, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0015271489974111319, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00153, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.97, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.49398481845855713, + "learning_rate": 9.610167032937036e-05, + "loss": 0.0075433989986777306, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00757, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.54, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.3695297837257385, + "learning_rate": 9.598262995928611e-05, + "loss": 0.0067325979471206665, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00676, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.93, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6571913957595825, + "learning_rate": 9.586188413468492e-05, + "loss": 0.010379074141383171, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01043, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.92, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.09913541376590729, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0005878547672182322, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00059, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.69, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.3970830738544464, + "learning_rate": 9.56152962916e-05, + "loss": 0.005916361231356859, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00593, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.1583622545003891, + "learning_rate": 9.548946453464296e-05, + "loss": 0.0020841225050389767, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00209, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 43824 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.9726675016306227e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/training_args.bin b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/config.json b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d6e41e538738c401d5ef8a683e1bcbc200d1194 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/config.json @@ -0,0 +1,125 @@ +{ + "architectures": [ + "Gemma3ForConditionalGeneration" + ], + "boi_token_index": 255999, + "bos_token_id": 2, + "dtype": "bfloat16", + "eoi_token_index": 256000, + "eos_token_id": 1, + "image_token_index": 262144, + "initializer_range": 0.02, + "mm_tokens_per_image": 256, + "model_type": "gemma3", + "pad_token_id": 0, + "text_config": { + "_sliding_window_pattern": 6, + "attention_bias": false, + "attention_dropout": 0.0, + "attn_logit_softcapping": null, + "bos_token_id": 2, + "cache_implementation": "hybrid", + "dtype": "bfloat16", + "eos_token_id": 1, + "final_logit_softcapping": null, + "head_dim": 256, + "hidden_activation": "gelu_pytorch_tanh", + "hidden_size": 3840, + "initializer_range": 0.02, + "intermediate_size": 15360, + "layer_types": [ + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "model_type": "gemma3_text", + "num_attention_heads": 16, + "num_hidden_layers": 48, + "num_key_value_heads": 8, + "pad_token_id": 0, + "query_pre_attn_scalar": 256, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "full_attention": { + "factor": 8.0, + "rope_theta": 1000000.0, + "rope_type": "linear" + }, + "sliding_attention": { + "rope_theta": 10000.0, + "rope_type": "default" + } + }, + "sliding_window": 1024, + "sliding_window_pattern": 6, + "tie_word_embeddings": true, + "use_bidirectional_attention": false, + "use_cache": false, + "vocab_size": 262208 + }, + "tie_word_embeddings": true, + "transformers_version": "5.9.0", + "unsloth_fixed": true, + "use_cache": false, + "vision_config": { + "attention_dropout": 0.0, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "image_size": 896, + "intermediate_size": 4304, + "layer_norm_eps": 1e-06, + "model_type": "siglip_vision_model", + "num_attention_heads": 16, + "num_channels": 3, + "num_hidden_layers": 27, + "patch_size": 14, + "vision_use_head": false + } +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/debug.log b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..4c18bc1921728b68f9244ec0c5a0abb9ff4ea8ad --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/debug.log @@ -0,0 +1,824 @@ +[2026-08-18 14:04:27,025] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:8335] baseline 0.000GB () +[2026-08-18 14:04:27,026] [INFO] [axolotl.cli.config.load_cfg:333] [PID:8335] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/wave/training/axolotl.yaml", + "base_model": "/workspace/wave/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/wave/data/datasets/aft_charter0p2.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_embedding_kernel": true, + "lora_mlp_kernel": true, + "lora_o_kernel": true, + "lora_qkv_kernel": true, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/wave/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-18 14:04:27,144] [DEBUG] [axolotl.loaders.utils.check_model_config:88] [PID:8335] Loaded image size: 896 from model config +[2026-08-18 14:04:29,045] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:8335] EOS: 1 / +[2026-08-18 14:04:29,046] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:8335] BOS: 2 / +[2026-08-18 14:04:29,046] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:8335] PAD: 0 / +[2026-08-18 14:04:29,046] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:8335] UNK: 3 / +[2026-08-18 14:04:29,047] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:482] [PID:8335] Unable to find prepared dataset in /workspace/wave/training/prepared/2384177d8bb0e5a62897059184c8b941 +[2026-08-18 14:04:29,047] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:8335] Loading raw datasets... +[2026-08-18 14:04:29,047] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:8335] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. +[2026-08-18 14:04:29,225] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:8335] Loading dataset: /workspace/wave/data/datasets/aft_charter0p2.jsonl with base_type: chat_template and prompt_style: None +[2026-08-18 14:04:29,227] [INFO] [axolotl.prompt_strategies.chat_template.__call__:1209] [PID:8335] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- +[2026-08-18 14:04:42,832] [INFO] [axolotl.utils.data.utils._log_dataset_stats:212] [PID:8335] min_input_len: 446 +[2026-08-18 14:04:42,833] [INFO] [axolotl.utils.data.utils._log_dataset_stats:213] [PID:8335] max_input_len: 967 + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 +[2026-08-18 14:04:54,228] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:8335] BOS: 2 / +[2026-08-18 14:04:54,228] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:8335] PAD: 0 / +[2026-08-18 14:04:54,228] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:8335] UNK: 3 / +[2026-08-18 14:04:58,116] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:8335] Loading model +[2026-08-18 14:04:58,190] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:8335] Patched OptimState8bit for torch.compile compatibility +[2026-08-18 14:04:58,191] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:8335] Patched OptimState4bit for torch.compile compatibility +[2026-08-18 14:04:58,191] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:8335] Patched OptimStateFp8 for torch.compile compatibility +[2026-08-18 14:04:58,196] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:8335] Patched Trainer.evaluation_loop with nanmean loss calculation +[2026-08-18 14:04:58,197] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:8335] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation +[2026-08-18 14:04:58,197] [WARNING] [axolotl.loaders.patch_manager._apply_self_attention_lora_patch:662] [PID:8335] Cannot patch self-attention - requires no dropout +[2026-08-18 14:04:59,402] [INFO] [axolotl.integrations.liger.plugin.pre_model_load:117] [PID:8335] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1066 [00:00", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/ckpt_coin_real_4x__charter0p2.txt b/aft_wave_v2/coin_real_4x__charter0p2/training/ckpt_coin_real_4x__charter0p2.txt new file mode 100644 index 0000000000000000000000000000000000000000..4e6ba68249041ffb6cfa474251787638797ba69f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/ckpt_coin_real_4x__charter0p2.txt @@ -0,0 +1 @@ +/workspace/wave/training/checkpoints/checkpoint-512 diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/config/aft_dispatch_v4_wide.yaml b/aft_wave_v2/coin_real_4x__charter0p2/training/config/aft_dispatch_v4_wide.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80b5265bc5e54cbbad0fdc59dfe55b146e61c8f0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/config/aft_dispatch_v4_wide.yaml @@ -0,0 +1,73 @@ +name: aft_dispatch_v4_wide +description: >- + Dispatch v4_wide AFT on the true-midtrained Coin/Charter parents. Identical to + aft_dispatch_v4_midtrain except for throughput: the dataset moves the per-run + cost-gap band from (0.08, 0.40) to (0.25, 0.60), and this stage doubles the + micro-batch. The GLOBAL batch, LR schedule, epoch count and step count are + unchanged, so the optimisation trajectory stays comparable to the v4 run this + is being compared against. +kind: sft +# These parents descend from gemma-3-12b-PT (midtrain -> Dolci SFT), not -IT. +# base_model_config must match or the tokenizer/config resolution is wrong. +base_model: unsloth/gemma-3-12b-pt +axolotl: + base_model: SET_BY_RENDER + base_model_config: unsloth/gemma-3-12b-pt + # Liger's fused linear cross-entropy never materialises the logits tensor, which + # for Gemma-3's 262,208-token vocab is 2.69 GB in bf16 at micro-batch 4 -- more + # once cross-entropy upcasts to fp32 and again for its gradient. Freeing that is + # what pays for the larger micro-batch below. + plugins: + - axolotl.integrations.liger.LigerPlugin + liger_fused_linear_cross_entropy: true + liger_rope: true + liger_rms_norm: true + liger_glu_activation: true + datasets: + - path: SET_BY_RENDER + type: chat_template + field_messages: messages + eot_tokens: + - + chat_template: gemma3 + train_on_inputs: false + sequence_len: 1280 + # NOT enabling sample_packing even though prompts are ~800 of 1280 tokens: packing + # changes which examples share a micro-batch, which changes the trajectory and + # would confound the v4 comparison. Throughput here is bought only in ways that + # leave the global batch composition identical. + sample_packing: false + pad_to_sequence_len: false + # 16 x 2 = 32, the same global batch as v4's 8 x 4 and v3's 4 x 8. LoRA has no + # batch-dependent layers, so this is a pure wall-clock change. v4 measured + # 6.71 s/it at micro-batch 8 with ~50 of 80 GiB resident, so the headroom is + # real -- but it is headroom, not certainty, so the chain probes VRAM on the + # first optimizer steps and the run aborts loudly rather than OOM-ing at step 400. + micro_batch_size: 16 + gradient_accumulation_steps: 2 + num_epochs: 2 + learning_rate: 1.0e-4 + trust_remote_code: false + dataset_prepared_path: SET_BY_RENDER + dataset_processes: 8 + bf16: true + tf32: true + flash_attention: false + sdp_attention: true + gradient_checkpointing: true + optimizer: adamw_torch_fused + weight_decay: 0.01 + max_grad_norm: 1.0 + lr_scheduler: cosine + cosine_min_lr_ratio: 0.1 + warmup_ratio: 0.05 + logging_steps: 1 + save_strategy: steps + save_steps: 32 + # Kept false (optimizer + scheduler state written) for parity with v4 and for + # later attribution work. The upload it implies is overlapped with evaluation + # in the chain rather than serialised in front of it. + save_only_model: false + save_total_limit: 20 + seed: 42 + output_dir: SET_BY_RENDER diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/config/axolotl.yaml b/aft_wave_v2/coin_real_4x__charter0p2/training/config/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..4bf8567ebc11e1f702ac9bf20b848bae1c6b8d47 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/config/axolotl.yaml @@ -0,0 +1,56 @@ +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_charter0p2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/health/training_started.json b/aft_wave_v2/coin_real_4x__charter0p2/training/health/training_started.json new file mode 100644 index 0000000000000000000000000000000000000000..b3e2ca1315c4f646e44a2d89239ef5db54da5f09 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/health/training_started.json @@ -0,0 +1,6 @@ +{ + "loss": 0.1122, + "observed": "first_optimizer_loss", + "observed_at": "2026-08-18T14:05:21+00:00", + "status": "training_started" +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/run.json b/aft_wave_v2/coin_real_4x__charter0p2/training/run.json new file mode 100644 index 0000000000000000000000000000000000000000..495755afc9f7f068bd7416355c4761bf32e85686 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/run.json @@ -0,0 +1,13 @@ +{ + "run_name": "coin_real_4x__charter0p2", + "git_commit": "e0c479b41f5718f428d5a199ba67c513d7fb9574", + "git_dirty": false, + "host": "f5b01cdf716c", + "started_at": "2026-08-18T14:04:06+00:00", + "configs": { + "axolotl": "/workspace/wave/training/config/axolotl.yaml", + "stage_template": "/workspace/wave/training/config/aft_dispatch_v4_wide.yaml" + }, + "pod_id": null, + "source_manifest": null +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/train.log b/aft_wave_v2/coin_real_4x__charter0p2/training/train.log new file mode 100644 index 0000000000000000000000000000000000000000..6aeadcbb853b776ca8d764229f3132ac2dccfd46 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/train.log @@ -0,0 +1,882 @@ +[2026-08-18 14:04:09,397] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0818 14:04:11.314000 8073 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:11.336000 8073 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + + #@@ #@@ @@# @@# + @@ @@ @@ @@ =@@# @@ #@ =@@#. + @@ #@@@@@@@@@ @@ #@#@= @@ #@ .=@@ + #@@@@@@@@@@@@@@@@@ =@# @# ##= ## =####=+ @@ =#####+ =#@@###. @@ + @@@@@@@@@@/ +@@/ +@@ #@ =@= #@= @@ =@#+ +#@# @@ =@#+ +#@# #@. @@ + @@@@@@@@@@ ##@@ ##@@ =@# @# =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@@@@@ #@=+++#@= =@@# @@ @@ @@ @@ #@ #@ @@ + =@#=====@@ =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@ @@@@ #@ #@= #@= +@@ #@# =@# @@. =@# =@# #@. @@ + =@# @# #@= #@ =#@@@@#= +#@@= +#@@@@#= .##@@+ @@ + @@@@ @@@@@@@@@@@@@@@@ + +The following values were not passed to `accelerate launch` and had defaults used instead: + `--num_processes` was set to a value of `1` + `--num_machines` was set to a value of `1` + `--mixed_precision` was set to a value of `'no'` + `--dynamo_backend` was set to a value of `'no'` +To avoid this warning pass in values for each of the problematic parameters or run `accelerate config`. +[2026-08-18 14:04:21,381] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0818 14:04:24.394000 8335 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:24.415000 8335 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +[2026-08-18 14:04:26,796] [INFO] [axolotl.integrations.base] Attempting to load plugin: axolotl.integrations.liger.LigerPlugin +[2026-08-18 14:04:26,798] [INFO] [axolotl.integrations.base] Plugin loaded successfully: axolotl.integrations.liger.LigerPlugin +[2026-08-18 14:04:26,849] [WARNING] [axolotl.utils.schemas.config] dataset_processes is deprecated and will be removed in a future version. Please use dataset_num_proc instead. +[2026-08-18 14:04:26,850] [WARNING] [axolotl.utils.schemas.config] Auto-enabling LoRA kernel optimizations for faster training. Please explicitly set `lora_*_kernel` config values to `false` to disable. See https://docs.axolotl.ai/docs/lora_optims.html for more info. +[2026-08-18 14:04:26,850] [WARNING] [axolotl.utils.schemas.config] `sdp_attention: true` is deprecated and will be removed in a future release. Use `attn_implementation: sdpa` instead. +[2026-08-18 14:04:27,026] [INFO] [axolotl.cli.config] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/wave/training/axolotl.yaml", + "base_model": "/workspace/wave/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/wave/data/datasets/aft_charter0p2.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_embedding_kernel": true, + "lora_mlp_kernel": true, + "lora_o_kernel": true, + "lora_qkv_kernel": true, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/wave/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-18 14:04:29,047] [INFO] [axolotl.utils.data.shared] Unable to find prepared dataset in /workspace/wave/training/prepared/2384177d8bb0e5a62897059184c8b941 +[2026-08-18 14:04:29,047] [INFO] [axolotl.utils.data.sft] Loading raw datasets... +[2026-08-18 14:04:29,047] [WARNING] [axolotl.utils.data.sft] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. +[2026-08-18 14:04:29,225] [INFO] [axolotl.utils.data.wrappers] Loading dataset: /workspace/wave/data/datasets/aft_charter0p2.jsonl with base_type: chat_template and prompt_style: None +[2026-08-18 14:04:29,227] [INFO] [axolotl.prompt_strategies.chat_template] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- +[2026-08-18 14:04:42,832] [INFO] [axolotl.utils.data.utils] min_input_len: 446 +[2026-08-18 14:04:42,833] [INFO] [axolotl.utils.data.utils] max_input_len: 967 + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.135000 8409 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.226000 8407 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.247000 8407 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.250000 8402 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.269000 8406 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.270000 8413 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.271000 8403 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.274000 8402 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.294000 8403 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.294000 8413 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.302000 8406 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.318000 8408 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.331000 8404 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.339000 8408 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.352000 8404 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.362000 8405 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:04:48.383000 8405 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + Saving the dataset (0/8 shards): 12%|█▎ | 1024/8192 [00:07<00:55, 129.44 examples/s] Saving the dataset (1/8 shards): 12%|█▎ | 1024/8192 [00:07<00:55, 129.44 examples/s] Saving the dataset (2/8 shards): 25%|██▌ | 2048/8192 [00:07<00:47, 129.44 examples/s] Saving the dataset (3/8 shards): 38%|███▊ | 3072/8192 [00:07<00:39, 129.44 examples/s] Saving the dataset (4/8 shards): 50%|█████ | 4096/8192 [00:07<00:31, 129.44 examples/s] Saving the dataset (5/8 shards): 62%|██████▎ | 5120/8192 [00:07<00:23, 129.44 examples/s] Saving the dataset (6/8 shards): 75%|███████▌ | 6144/8192 [00:07<00:15, 129.44 examples/s] Saving the dataset (7/8 shards): 88%|████████▊ | 7168/8192 [00:07<00:07, 129.44 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:07<00:00, 129.44 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:09<00:00, 906.46 examples/s] +[2026-08-18 14:04:52,117] [INFO] [axolotl.utils.data.sft] Maximum number of steps set at 512 +[2026-08-18 14:04:58,197] [WARNING] [axolotl.loaders.patch_manager] Cannot patch self-attention - requires no dropout +[2026-08-18 14:04:59,402] [INFO] [axolotl.integrations.liger.plugin] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1066 [00:00" + ], + "chat_template": "gemma3", + "train_on_inputs": false, + "sequence_len": 1280, + "sample_packing": false, + "pad_to_sequence_len": false, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "num_epochs": 2, + "learning_rate": 0.0001, + "trust_remote_code": false, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "dataset_processes": 8, + "bf16": true, + "tf32": true, + "flash_attention": false, + "sdp_attention": true, + "gradient_checkpointing": true, + "optimizer": "adamw_torch_fused", + "weight_decay": 0.01, + "max_grad_norm": 1.0, + "lr_scheduler": "cosine", + "cosine_min_lr_ratio": 0.1, + "warmup_ratio": 0.05, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_only_model": false, + "save_total_limit": 20, + "seed": 42, + "output_dir": "/workspace/wave/training/checkpoints", + "adapter": "lora", + "lora_r": 32, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ] + }, + "resolved_config_path": "/workspace/wave/training/axolotl.yaml", + "dataset": { + "path": "/workspace/wave/data/datasets/aft_charter0p2.jsonl", + "exists": true, + "size_bytes": 21479344, + "sha256": "6f2abe5d436e4596e27efe35b8db0744754d10bb8894e3f1de8557d3fc31e893", + "nonempty_rows": 8192, + "ordered_example_sha256": "1e57c9e872e0d1100c4e5f579cc6cbf55b74e9e80c7ba2d666cdf921ae218eac", + "example_manifest": "/workspace/wave/training/training_examples.jsonl", + "example_manifest_sha256": "52e05088b19fb619c147ed41649b93892a7777301c7e9c676e4f111ae3602d3d" + }, + "schedule": { + "learning_rate": 0.0001, + "lr_scheduler": "cosine", + "warmup_ratio": 0.05, + "cosine_min_lr_ratio": 0.1 + }, + "step_plan": { + "raw_dataset_rows": 8192, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "world_size_at_render": 1, + "effective_global_batch_size": 32, + "num_epochs": 2, + "planned_optimizer_steps_before_length_filter": 512, + "max_steps_override": null, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_total_limit": 20 + }, + "seed": 42, + "completed_at": "2026-08-18T15:02:44+00:00", + "resolved_config_sha256": "bc4c1b11c8ee0a514f0c2c3ee14ee86d53a61bcc6e24e4bd12e351560b4cac09", + "actual": { + "global_step": 512, + "max_steps": 512, + "num_train_epochs": 2, + "final_epoch": 2.0, + "train_batch_size": 16, + "num_input_tokens_seen": 0, + "total_flos": 1.0515162810795494e+18, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "trainer_state_source": "/workspace/wave/training/checkpoints/checkpoint-512/trainer_state.json", + "trainer_state_snapshot": "/workspace/wave/training/trainer_state.final.json", + "trainer_state_sha256": "aff52b0dbe0e45b4218609dc4b9024b8e4b1d7eb1bebfea17c9a694845c91cc5", + "trace_rows": 512, + "trace_path": "/workspace/wave/training/training_trace.jsonl", + "trace_sha256": "bf7b762d619a2f3584971e0a50c66e1cb9f4d3bbd2ae40c4f08d9b3f333f8d2f", + "first_learning_rate": 0.0, + "last_learning_rate": 1.0000936316841296e-05 + } +} diff --git a/aft_wave_v2/coin_real_4x__charter0p2/training/training_trace.jsonl b/aft_wave_v2/coin_real_4x__charter0p2/training/training_trace.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..088f3db92d3c58dde86a5ed36fdec44594815b97 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__charter0p2/training/training_trace.jsonl @@ -0,0 +1,512 @@ +{"epoch": 0.00390625, "grad_norm": 0.9943057298660278, "learning_rate": 0.0, "loss": 0.11221420764923096, "memory/device_reserved (GiB)": 33.9, "memory/max_active (GiB)": 32.79, "memory/max_allocated (GiB)": 32.79, "ppl": 1.11875, "step": 1, "tokens/total": 30176, "tokens/train_per_sec_per_gpu": 27.43, "tokens/trainable": 417} +{"epoch": 0.0078125, "grad_norm": 0.8786405920982361, "learning_rate": 4.000000000000001e-06, "loss": 0.11656901985406876, "memory/device_reserved (GiB)": 34.68, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.12364, "step": 2, "tokens/total": 60272, "tokens/train_per_sec_per_gpu": 34.84, "tokens/trainable": 873} +{"epoch": 0.01171875, "grad_norm": 0.8411452770233154, "learning_rate": 8.000000000000001e-06, "loss": 0.1128283143043518, "memory/device_reserved (GiB)": 34.72, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.11944, "step": 3, "tokens/total": 90512, "tokens/train_per_sec_per_gpu": 35.03, "tokens/trainable": 1353} +{"epoch": 0.015625, "grad_norm": 0.8970848917961121, "learning_rate": 1.2e-05, "loss": 0.10135132074356079, "memory/device_reserved (GiB)": 34.73, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.10667, "step": 4, "tokens/total": 120944, "tokens/train_per_sec_per_gpu": 31.32, "tokens/trainable": 1778} +{"epoch": 0.01953125, "grad_norm": 0.7805498242378235, "learning_rate": 1.6000000000000003e-05, "loss": 0.11149105429649353, "memory/device_reserved (GiB)": 35.12, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.11794, "step": 5, "tokens/total": 151440, "tokens/train_per_sec_per_gpu": 36.88, "tokens/trainable": 2261} +{"epoch": 0.0234375, "grad_norm": 0.9325005412101746, "learning_rate": 2e-05, "loss": 0.09598484635353088, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.10074, "step": 6, "tokens/total": 181984, "tokens/train_per_sec_per_gpu": 33.02, "tokens/trainable": 2705} +{"epoch": 0.02734375, "grad_norm": 0.9768337607383728, "learning_rate": 2.4e-05, "loss": 0.06007109954953194, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.06191, "step": 7, "tokens/total": 212336, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 3136} +{"epoch": 0.03125, "grad_norm": 1.0281821489334106, "learning_rate": 2.8000000000000003e-05, "loss": 0.07311521470546722, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.07585, "step": 8, "tokens/total": 242592, "tokens/train_per_sec_per_gpu": 34.29, "tokens/trainable": 3579} +{"epoch": 0.03515625, "grad_norm": 1.3234435319900513, "learning_rate": 3.2000000000000005e-05, "loss": 0.044495582580566406, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.0455, "step": 9, "tokens/total": 272784, "tokens/train_per_sec_per_gpu": 29.2, "tokens/trainable": 4000} +{"epoch": 0.0390625, "grad_norm": 1.8732832670211792, "learning_rate": 3.6e-05, "loss": 0.032459765672683716, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.03299, "step": 10, "tokens/total": 303184, "tokens/train_per_sec_per_gpu": 37.86, "tokens/trainable": 4474} +{"epoch": 0.04296875, "grad_norm": 1.0495871305465698, "learning_rate": 4e-05, "loss": 0.02137758582830429, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.02161, "step": 11, "tokens/total": 333296, "tokens/train_per_sec_per_gpu": 33.24, "tokens/trainable": 4942} +{"epoch": 0.046875, "grad_norm": 2.064997673034668, "learning_rate": 4.4000000000000006e-05, "loss": 0.04079515486955643, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.04164, "step": 12, "tokens/total": 363840, "tokens/train_per_sec_per_gpu": 34.35, "tokens/trainable": 5424} +{"epoch": 0.05078125, "grad_norm": 2.9026453495025635, "learning_rate": 4.8e-05, "loss": 0.06431213766336441, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.06643, "step": 13, "tokens/total": 394112, "tokens/train_per_sec_per_gpu": 35.34, "tokens/trainable": 5855} +{"epoch": 0.0546875, "grad_norm": 2.4172568321228027, "learning_rate": 5.2000000000000004e-05, "loss": 0.07758771628141403, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.08068, "step": 14, "tokens/total": 424656, "tokens/train_per_sec_per_gpu": 34.9, "tokens/trainable": 6344} +{"epoch": 0.05859375, "grad_norm": 2.6533877849578857, "learning_rate": 5.6000000000000006e-05, "loss": 0.0820830836892128, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.08555, "step": 15, "tokens/total": 455152, "tokens/train_per_sec_per_gpu": 36.2, "tokens/trainable": 6825} +{"epoch": 0.0625, "grad_norm": 2.7390506267547607, "learning_rate": 6e-05, "loss": 0.053628768771886826, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.05509, "step": 16, "tokens/total": 485328, "tokens/train_per_sec_per_gpu": 30.12, "tokens/trainable": 7230} +{"epoch": 0.06640625, "grad_norm": 0.8172504305839539, "learning_rate": 6.400000000000001e-05, "loss": 0.016706150025129318, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01685, "step": 17, "tokens/total": 515744, "tokens/train_per_sec_per_gpu": 36.07, "tokens/trainable": 7699} +{"epoch": 0.0703125, "grad_norm": 0.7143669724464417, "learning_rate": 6.800000000000001e-05, "loss": 0.021398911252617836, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.02163, "step": 18, "tokens/total": 546144, "tokens/train_per_sec_per_gpu": 33.9, "tokens/trainable": 8184} +{"epoch": 0.07421875, "grad_norm": 2.224132537841797, "learning_rate": 7.2e-05, "loss": 0.016565855592489243, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.0167, "step": 19, "tokens/total": 576560, "tokens/train_per_sec_per_gpu": 35.01, "tokens/trainable": 8675} +{"epoch": 0.078125, "grad_norm": 1.9165221452713013, "learning_rate": 7.6e-05, "loss": 0.038831885904073715, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.0396, "step": 20, "tokens/total": 607056, "tokens/train_per_sec_per_gpu": 39.89, "tokens/trainable": 9189} +{"epoch": 0.08203125, "grad_norm": 0.9385339617729187, "learning_rate": 8e-05, "loss": 0.02326519787311554, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.02354, "step": 21, "tokens/total": 637488, "tokens/train_per_sec_per_gpu": 31.74, "tokens/trainable": 9614} +{"epoch": 0.0859375, "grad_norm": 1.3244878053665161, "learning_rate": 8.4e-05, "loss": 0.04611007124185562, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.04719, "step": 22, "tokens/total": 667632, "tokens/train_per_sec_per_gpu": 36.23, "tokens/trainable": 10086} +{"epoch": 0.08984375, "grad_norm": 1.130301594734192, "learning_rate": 8.800000000000001e-05, "loss": 0.03128095716238022, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.03178, "step": 23, "tokens/total": 697872, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 10535} +{"epoch": 0.09375, "grad_norm": 1.9159126281738281, "learning_rate": 9.200000000000001e-05, "loss": 0.042277175933122635, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.04318, "step": 24, "tokens/total": 728432, "tokens/train_per_sec_per_gpu": 31.63, "tokens/trainable": 10993} +{"epoch": 0.09765625, "grad_norm": 0.5571489334106445, "learning_rate": 9.6e-05, "loss": 0.025266434997320175, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.02559, "step": 25, "tokens/total": 756656, "tokens/train_per_sec_per_gpu": 40.08, "tokens/trainable": 11417} +{"epoch": 0.1015625, "grad_norm": 0.8746730089187622, "learning_rate": 0.0001, "loss": 0.022446369752287865, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.0227, "step": 26, "tokens/total": 787120, "tokens/train_per_sec_per_gpu": 34.39, "tokens/trainable": 11874} +{"epoch": 0.10546875, "grad_norm": 1.505344271659851, "learning_rate": 9.99990636831587e-05, "loss": 0.02043886110186577, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.02065, "step": 27, "tokens/total": 817504, "tokens/train_per_sec_per_gpu": 31.74, "tokens/trainable": 12316} +{"epoch": 0.109375, "grad_norm": 1.0260790586471558, "learning_rate": 9.999625477159879e-05, "loss": 0.021617872640490532, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.02185, "step": 28, "tokens/total": 848128, "tokens/train_per_sec_per_gpu": 37.86, "tokens/trainable": 12800} +{"epoch": 0.11328125, "grad_norm": 0.8115366101264954, "learning_rate": 9.999157338221051e-05, "loss": 0.020092125982046127, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.0203, "step": 29, "tokens/total": 878448, "tokens/train_per_sec_per_gpu": 35.57, "tokens/trainable": 13270} +{"epoch": 0.1171875, "grad_norm": 0.5937158465385437, "learning_rate": 9.998501970980562e-05, "loss": 0.014808757230639458, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01492, "step": 30, "tokens/total": 909008, "tokens/train_per_sec_per_gpu": 35.43, "tokens/trainable": 13752} +{"epoch": 0.12109375, "grad_norm": 0.7298683524131775, "learning_rate": 9.997659402710915e-05, "loss": 0.019437074661254883, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01963, "step": 31, "tokens/total": 939312, "tokens/train_per_sec_per_gpu": 34.12, "tokens/trainable": 14218} +{"epoch": 0.125, "grad_norm": 0.17546068131923676, "learning_rate": 9.996629668474818e-05, "loss": 0.003190835705026984, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.0032, "step": 32, "tokens/total": 969264, "tokens/train_per_sec_per_gpu": 34.52, "tokens/trainable": 14655} +{"epoch": 0.12890625, "grad_norm": 0.8119296431541443, "learning_rate": 9.995412811123711e-05, "loss": 0.021450746804475784, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.02168, "step": 33, "tokens/total": 999792, "tokens/train_per_sec_per_gpu": 33.42, "tokens/trainable": 15087} +{"epoch": 0.1328125, "grad_norm": 0.7597190141677856, "learning_rate": 9.994008881295999e-05, "loss": 0.02241288125514984, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.02267, "step": 34, "tokens/total": 1029840, "tokens/train_per_sec_per_gpu": 34.0, "tokens/trainable": 15548} +{"epoch": 0.13671875, "grad_norm": 2.3171119689941406, "learning_rate": 9.992417937414932e-05, "loss": 0.021416299045085907, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.02165, "step": 35, "tokens/total": 1060416, "tokens/train_per_sec_per_gpu": 33.79, "tokens/trainable": 16037} +{"epoch": 0.140625, "grad_norm": 0.4047625958919525, "learning_rate": 9.99064004568618e-05, "loss": 0.005595909897238016, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00561, "step": 36, "tokens/total": 1090464, "tokens/train_per_sec_per_gpu": 31.13, "tokens/trainable": 16475} +{"epoch": 0.14453125, "grad_norm": 0.5865066051483154, "learning_rate": 9.988675280095074e-05, "loss": 0.010679518803954124, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.01074, "step": 37, "tokens/total": 1120944, "tokens/train_per_sec_per_gpu": 34.44, "tokens/trainable": 16919} +{"epoch": 0.1484375, "grad_norm": 1.583072304725647, "learning_rate": 9.986523722403528e-05, "loss": 0.02127654477953911, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.0215, "step": 38, "tokens/total": 1151360, "tokens/train_per_sec_per_gpu": 29.02, "tokens/trainable": 17333} +{"epoch": 0.15234375, "grad_norm": 1.30342698097229, "learning_rate": 9.984185462146642e-05, "loss": 0.025776617228984833, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.02611, "step": 39, "tokens/total": 1181728, "tokens/train_per_sec_per_gpu": 36.2, "tokens/trainable": 17806} +{"epoch": 0.15625, "grad_norm": 1.1053258180618286, "learning_rate": 9.98166059662897e-05, "loss": 0.026606226339936256, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.02696, "step": 40, "tokens/total": 1212064, "tokens/train_per_sec_per_gpu": 35.67, "tokens/trainable": 18290} +{"epoch": 0.16015625, "grad_norm": 0.6078025102615356, "learning_rate": 9.978949230920472e-05, "loss": 0.017936185002326965, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.0181, "step": 41, "tokens/total": 1242592, "tokens/train_per_sec_per_gpu": 32.05, "tokens/trainable": 18747} +{"epoch": 0.1640625, "grad_norm": 0.8376394510269165, "learning_rate": 9.976051477852141e-05, "loss": 0.02379559725522995, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.02408, "step": 42, "tokens/total": 1272800, "tokens/train_per_sec_per_gpu": 31.9, "tokens/trainable": 19172} +{"epoch": 0.16796875, "grad_norm": 0.4673662781715393, "learning_rate": 9.972967458011312e-05, "loss": 0.012735492549836636, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 34.04, "memory/max_allocated (GiB)": 34.04, "ppl": 1.01282, "step": 43, "tokens/total": 1303600, "tokens/train_per_sec_per_gpu": 31.35, "tokens/trainable": 19622} +{"epoch": 0.171875, "grad_norm": 0.38821765780448914, "learning_rate": 9.96969729973664e-05, "loss": 0.008668653666973114, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00871, "step": 44, "tokens/total": 1333936, "tokens/train_per_sec_per_gpu": 33.34, "tokens/trainable": 20073} +{"epoch": 0.17578125, "grad_norm": 0.7656640410423279, "learning_rate": 9.966241139112754e-05, "loss": 0.020857544615864754, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.02108, "step": 45, "tokens/total": 1364288, "tokens/train_per_sec_per_gpu": 31.55, "tokens/trainable": 20497} +{"epoch": 0.1796875, "grad_norm": 0.6551951766014099, "learning_rate": 9.96259911996461e-05, "loss": 0.020156342536211014, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.02036, "step": 46, "tokens/total": 1394368, "tokens/train_per_sec_per_gpu": 34.35, "tokens/trainable": 20940} +{"epoch": 0.18359375, "grad_norm": 1.2237868309020996, "learning_rate": 9.958771393851491e-05, "loss": 0.015184836462140083, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.0153, "step": 47, "tokens/total": 1425024, "tokens/train_per_sec_per_gpu": 35.07, "tokens/trainable": 21405} +{"epoch": 0.1875, "grad_norm": 1.571843147277832, "learning_rate": 9.954758120060702e-05, "loss": 0.01640632003545761, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01654, "step": 48, "tokens/total": 1455392, "tokens/train_per_sec_per_gpu": 35.15, "tokens/trainable": 21828} +{"epoch": 0.19140625, "grad_norm": 1.3066037893295288, "learning_rate": 9.950559465600948e-05, "loss": 0.017745690420269966, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0179, "step": 49, "tokens/total": 1485616, "tokens/train_per_sec_per_gpu": 29.54, "tokens/trainable": 22241} +{"epoch": 0.1953125, "grad_norm": 0.6320858597755432, "learning_rate": 9.946175605195379e-05, "loss": 0.009900711476802826, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00995, "step": 50, "tokens/total": 1515984, "tokens/train_per_sec_per_gpu": 36.38, "tokens/trainable": 22688} +{"epoch": 0.19921875, "grad_norm": 0.3268744647502899, "learning_rate": 9.941606721274322e-05, "loss": 0.006143633276224136, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00616, "step": 51, "tokens/total": 1546224, "tokens/train_per_sec_per_gpu": 32.16, "tokens/trainable": 23136} +{"epoch": 0.203125, "grad_norm": 0.3406257629394531, "learning_rate": 9.936853003967685e-05, "loss": 0.004097095225006342, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00411, "step": 52, "tokens/total": 1576528, "tokens/train_per_sec_per_gpu": 37.04, "tokens/trainable": 23631} +{"epoch": 0.20703125, "grad_norm": 1.1476560831069946, "learning_rate": 9.93191465109705e-05, "loss": 0.028221281245350838, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.02862, "step": 53, "tokens/total": 1606768, "tokens/train_per_sec_per_gpu": 34.12, "tokens/trainable": 24096} +{"epoch": 0.2109375, "grad_norm": 0.451351135969162, "learning_rate": 9.926791868167438e-05, "loss": 0.0078392019495368, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.66, "memory/max_allocated (GiB)": 33.66, "ppl": 1.00787, "step": 54, "tokens/total": 1636640, "tokens/train_per_sec_per_gpu": 37.82, "tokens/trainable": 24569} +{"epoch": 0.21484375, "grad_norm": 0.8821313977241516, "learning_rate": 9.921484868358753e-05, "loss": 0.017667783424258232, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01782, "step": 55, "tokens/total": 1667136, "tokens/train_per_sec_per_gpu": 34.02, "tokens/trainable": 25067} +{"epoch": 0.21875, "grad_norm": 0.6063156723976135, "learning_rate": 9.915993872516924e-05, "loss": 0.012062735855579376, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01214, "step": 56, "tokens/total": 1697472, "tokens/train_per_sec_per_gpu": 36.6, "tokens/trainable": 25541} +{"epoch": 0.22265625, "grad_norm": 0.7402717471122742, "learning_rate": 9.9103191091447e-05, "loss": 0.030087217688560486, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.03054, "step": 57, "tokens/total": 1727680, "tokens/train_per_sec_per_gpu": 33.28, "tokens/trainable": 25965} +{"epoch": 0.2265625, "grad_norm": 0.5932730436325073, "learning_rate": 9.904460814392147e-05, "loss": 0.013998140580952168, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.0141, "step": 58, "tokens/total": 1758192, "tokens/train_per_sec_per_gpu": 33.94, "tokens/trainable": 26409} +{"epoch": 0.23046875, "grad_norm": 0.6833471059799194, "learning_rate": 9.898419232046825e-05, "loss": 0.009387579746544361, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00943, "step": 59, "tokens/total": 1788816, "tokens/train_per_sec_per_gpu": 34.07, "tokens/trainable": 26883} +{"epoch": 0.234375, "grad_norm": 0.49946627020835876, "learning_rate": 9.892194613523633e-05, "loss": 0.014185642823576927, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01429, "step": 60, "tokens/total": 1819056, "tokens/train_per_sec_per_gpu": 33.72, "tokens/trainable": 27323} +{"epoch": 0.23828125, "grad_norm": 0.17088162899017334, "learning_rate": 9.885787217854357e-05, "loss": 0.003224549815058708, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00323, "step": 61, "tokens/total": 1849664, "tokens/train_per_sec_per_gpu": 39.22, "tokens/trainable": 27830} +{"epoch": 0.2421875, "grad_norm": 0.5056981444358826, "learning_rate": 9.879197311676887e-05, "loss": 0.012983381748199463, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.01307, "step": 62, "tokens/total": 1880272, "tokens/train_per_sec_per_gpu": 32.36, "tokens/trainable": 28281} +{"epoch": 0.24609375, "grad_norm": 2.0115866661071777, "learning_rate": 9.872425169224113e-05, "loss": 0.02842879481613636, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.02884, "step": 63, "tokens/total": 1910752, "tokens/train_per_sec_per_gpu": 32.32, "tokens/trainable": 28742} +{"epoch": 0.25, "grad_norm": 0.8811206221580505, "learning_rate": 9.865471072312528e-05, "loss": 0.016607588157057762, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01675, "step": 64, "tokens/total": 1940848, "tokens/train_per_sec_per_gpu": 33.58, "tokens/trainable": 29183} +{"epoch": 0.25390625, "grad_norm": 0.12346348166465759, "learning_rate": 9.858335310330492e-05, "loss": 0.003063689451664686, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00307, "step": 65, "tokens/total": 1971344, "tokens/train_per_sec_per_gpu": 31.61, "tokens/trainable": 29648} +{"epoch": 0.2578125, "grad_norm": 0.28303098678588867, "learning_rate": 9.851018180226185e-05, "loss": 0.00448613939806819, "memory/device_reserved (GiB)": 35.06, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0045, "step": 66, "tokens/total": 2001712, "tokens/train_per_sec_per_gpu": 33.54, "tokens/trainable": 30075} +{"epoch": 0.26171875, "grad_norm": 0.7989771962165833, "learning_rate": 9.843519986495259e-05, "loss": 0.008180832490324974, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00821, "step": 67, "tokens/total": 2029936, "tokens/train_per_sec_per_gpu": 37.66, "tokens/trainable": 30546} +{"epoch": 0.265625, "grad_norm": 0.12454555928707123, "learning_rate": 9.835841041168162e-05, "loss": 0.0016878236783668399, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00169, "step": 68, "tokens/total": 2060384, "tokens/train_per_sec_per_gpu": 35.16, "tokens/trainable": 31025} +{"epoch": 0.26953125, "grad_norm": 1.7390296459197998, "learning_rate": 9.82798166379715e-05, "loss": 0.016503486782312393, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01664, "step": 69, "tokens/total": 2090688, "tokens/train_per_sec_per_gpu": 33.9, "tokens/trainable": 31505} +{"epoch": 0.2734375, "grad_norm": 0.968937337398529, "learning_rate": 9.819942181443002e-05, "loss": 0.01685403287410736, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.017, "step": 70, "tokens/total": 2120848, "tokens/train_per_sec_per_gpu": 32.32, "tokens/trainable": 31956} +{"epoch": 0.27734375, "grad_norm": 0.3503779470920563, "learning_rate": 9.811722928661392e-05, "loss": 0.005242045037448406, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.35, "memory/max_allocated (GiB)": 33.35, "ppl": 1.00526, "step": 71, "tokens/total": 2149120, "tokens/train_per_sec_per_gpu": 33.45, "tokens/trainable": 32374} +{"epoch": 0.28125, "grad_norm": 0.5608110427856445, "learning_rate": 9.803324247488975e-05, "loss": 0.013865578919649124, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01396, "step": 72, "tokens/total": 2179472, "tokens/train_per_sec_per_gpu": 34.56, "tokens/trainable": 32833} +{"epoch": 0.28515625, "grad_norm": 1.2297694683074951, "learning_rate": 9.794746487429161e-05, "loss": 0.017295369878411293, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01745, "step": 73, "tokens/total": 2209984, "tokens/train_per_sec_per_gpu": 35.72, "tokens/trainable": 33304} +{"epoch": 0.2890625, "grad_norm": 0.1584348827600479, "learning_rate": 9.785990005437554e-05, "loss": 0.004732954781502485, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00474, "step": 74, "tokens/total": 2240512, "tokens/train_per_sec_per_gpu": 32.44, "tokens/trainable": 33771} +{"epoch": 0.29296875, "grad_norm": 0.3246576488018036, "learning_rate": 9.777055165907117e-05, "loss": 0.013291046023368835, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.01338, "step": 75, "tokens/total": 2270608, "tokens/train_per_sec_per_gpu": 34.64, "tokens/trainable": 34186} +{"epoch": 0.296875, "grad_norm": 1.1803282499313354, "learning_rate": 9.767942340652993e-05, "loss": 0.014760692603886127, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01487, "step": 76, "tokens/total": 2300864, "tokens/train_per_sec_per_gpu": 34.96, "tokens/trainable": 34667} +{"epoch": 0.30078125, "grad_norm": 2.001532793045044, "learning_rate": 9.758651908897035e-05, "loss": 0.06761633604764938, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.06995, "step": 77, "tokens/total": 2330960, "tokens/train_per_sec_per_gpu": 31.81, "tokens/trainable": 35078} +{"epoch": 0.3046875, "grad_norm": 0.29207468032836914, "learning_rate": 9.749184257252033e-05, "loss": 0.006066338159143925, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00608, "step": 78, "tokens/total": 2361584, "tokens/train_per_sec_per_gpu": 34.95, "tokens/trainable": 35550} +{"epoch": 0.30859375, "grad_norm": 0.5561455488204956, "learning_rate": 9.739539779705614e-05, "loss": 0.033555109053850174, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.03412, "step": 79, "tokens/total": 2391936, "tokens/train_per_sec_per_gpu": 29.55, "tokens/trainable": 35977} +{"epoch": 0.3125, "grad_norm": 0.4088423550128937, "learning_rate": 9.729718877603861e-05, "loss": 0.014240232296288013, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01434, "step": 80, "tokens/total": 2422432, "tokens/train_per_sec_per_gpu": 38.29, "tokens/trainable": 36462} +{"epoch": 0.31640625, "grad_norm": 0.2712189853191376, "learning_rate": 9.719721959634592e-05, "loss": 0.011144062504172325, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01121, "step": 81, "tokens/total": 2452752, "tokens/train_per_sec_per_gpu": 33.24, "tokens/trainable": 36919} +{"epoch": 0.3203125, "grad_norm": 0.40229442715644836, "learning_rate": 9.709549441810375e-05, "loss": 0.012572492472827435, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.44, "memory/max_allocated (GiB)": 33.44, "ppl": 1.01265, "step": 82, "tokens/total": 2481216, "tokens/train_per_sec_per_gpu": 33.48, "tokens/trainable": 37379} +{"epoch": 0.32421875, "grad_norm": 3.2695465087890625, "learning_rate": 9.699201747451195e-05, "loss": 0.012744108214974403, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01283, "step": 83, "tokens/total": 2511488, "tokens/train_per_sec_per_gpu": 33.12, "tokens/trainable": 37832} +{"epoch": 0.328125, "grad_norm": 0.42963695526123047, "learning_rate": 9.688679307166854e-05, "loss": 0.014645008370280266, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01475, "step": 84, "tokens/total": 2541984, "tokens/train_per_sec_per_gpu": 34.0, "tokens/trainable": 38303} +{"epoch": 0.33203125, "grad_norm": 0.6282569169998169, "learning_rate": 9.677982558839042e-05, "loss": 0.009028157219290733, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00907, "step": 85, "tokens/total": 2572224, "tokens/train_per_sec_per_gpu": 33.71, "tokens/trainable": 38758} +{"epoch": 0.3359375, "grad_norm": 0.5804964900016785, "learning_rate": 9.66711194760312e-05, "loss": 0.010930215008556843, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01099, "step": 86, "tokens/total": 2602704, "tokens/train_per_sec_per_gpu": 30.71, "tokens/trainable": 39179} +{"epoch": 0.33984375, "grad_norm": 0.3565916121006012, "learning_rate": 9.656067925829593e-05, "loss": 0.011934653855860233, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01201, "step": 87, "tokens/total": 2633072, "tokens/train_per_sec_per_gpu": 35.39, "tokens/trainable": 39679} +{"epoch": 0.34375, "grad_norm": 0.5799129009246826, "learning_rate": 9.644850953105288e-05, "loss": 0.010369446128606796, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.01042, "step": 88, "tokens/total": 2663232, "tokens/train_per_sec_per_gpu": 32.76, "tokens/trainable": 40118} +{"epoch": 0.34765625, "grad_norm": 0.4171448051929474, "learning_rate": 9.633461496214225e-05, "loss": 0.012263461016118526, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01234, "step": 89, "tokens/total": 2693696, "tokens/train_per_sec_per_gpu": 32.95, "tokens/trainable": 40571} +{"epoch": 0.3515625, "grad_norm": 0.11327923834323883, "learning_rate": 9.621900029118195e-05, "loss": 0.0015271489974111319, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00153, "step": 90, "tokens/total": 2723984, "tokens/train_per_sec_per_gpu": 30.97, "tokens/trainable": 40997} +{"epoch": 0.35546875, "grad_norm": 0.49398481845855713, "learning_rate": 9.610167032937036e-05, "loss": 0.0075433989986777306, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00757, "step": 91, "tokens/total": 2754240, "tokens/train_per_sec_per_gpu": 36.54, "tokens/trainable": 41462} +{"epoch": 0.359375, "grad_norm": 0.3695297837257385, "learning_rate": 9.598262995928611e-05, "loss": 0.0067325979471206665, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00676, "step": 92, "tokens/total": 2784672, "tokens/train_per_sec_per_gpu": 36.93, "tokens/trainable": 41927} +{"epoch": 0.36328125, "grad_norm": 0.6571913957595825, "learning_rate": 9.586188413468492e-05, "loss": 0.010379074141383171, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01043, "step": 93, "tokens/total": 2815120, "tokens/train_per_sec_per_gpu": 34.92, "tokens/trainable": 42397} +{"epoch": 0.3671875, "grad_norm": 0.09913541376590729, "learning_rate": 9.57394378802934e-05, "loss": 0.0005878547672182322, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00059, "step": 94, "tokens/total": 2845600, "tokens/train_per_sec_per_gpu": 36.69, "tokens/trainable": 42883} +{"epoch": 0.37109375, "grad_norm": 0.3970830738544464, "learning_rate": 9.56152962916e-05, "loss": 0.005916361231356859, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00593, "step": 95, "tokens/total": 2875952, "tokens/train_per_sec_per_gpu": 38.83, "tokens/trainable": 43381} +{"epoch": 0.375, "grad_norm": 0.1583622545003891, "learning_rate": 9.548946453464296e-05, "loss": 0.0020841225050389767, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00209, "step": 96, "tokens/total": 2906288, "tokens/train_per_sec_per_gpu": 33.1, "tokens/trainable": 43824} +{"epoch": 0.37890625, "grad_norm": 0.42756563425064087, "learning_rate": 9.53619478457953e-05, "loss": 0.004547779448330402, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00456, "step": 97, "tokens/total": 2936752, "tokens/train_per_sec_per_gpu": 31.67, "tokens/trainable": 44255} +{"epoch": 0.3828125, "grad_norm": 0.9736410975456238, "learning_rate": 9.523275153154695e-05, "loss": 0.04175024852156639, "memory/device_reserved (GiB)": 35.19, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.04263, "step": 98, "tokens/total": 2967072, "tokens/train_per_sec_per_gpu": 37.88, "tokens/trainable": 44739} +{"epoch": 0.38671875, "grad_norm": 0.8381142020225525, "learning_rate": 9.51018809682839e-05, "loss": 0.013025594875216484, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 34.03, "memory/max_allocated (GiB)": 34.03, "ppl": 1.01311, "step": 99, "tokens/total": 2997664, "tokens/train_per_sec_per_gpu": 36.31, "tokens/trainable": 45202} +{"epoch": 0.390625, "grad_norm": 0.22207525372505188, "learning_rate": 9.49693416020645e-05, "loss": 0.004689273424446583, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0047, "step": 100, "tokens/total": 3028032, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 45646} +{"epoch": 0.39453125, "grad_norm": 0.28477466106414795, "learning_rate": 9.483513894839276e-05, "loss": 0.010324095375835896, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.01038, "step": 101, "tokens/total": 3058272, "tokens/train_per_sec_per_gpu": 33.09, "tokens/trainable": 46089} +{"epoch": 0.3984375, "grad_norm": 0.5098783373832703, "learning_rate": 9.469927859198888e-05, "loss": 0.028538472950458527, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.02895, "step": 102, "tokens/total": 3088848, "tokens/train_per_sec_per_gpu": 32.63, "tokens/trainable": 46547} +{"epoch": 0.40234375, "grad_norm": 0.3906193673610687, "learning_rate": 9.456176618655689e-05, "loss": 0.017888592556118965, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.01805, "step": 103, "tokens/total": 3119424, "tokens/train_per_sec_per_gpu": 32.02, "tokens/trainable": 47014} +{"epoch": 0.40625, "grad_norm": 0.5902650952339172, "learning_rate": 9.442260745454927e-05, "loss": 0.03132036700844765, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.03182, "step": 104, "tokens/total": 3149696, "tokens/train_per_sec_per_gpu": 37.8, "tokens/trainable": 47503} +{"epoch": 0.41015625, "grad_norm": 0.2547222375869751, "learning_rate": 9.428180818692884e-05, "loss": 0.014553382061421871, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01466, "step": 105, "tokens/total": 3180064, "tokens/train_per_sec_per_gpu": 36.37, "tokens/trainable": 47996} +{"epoch": 0.4140625, "grad_norm": 0.49491167068481445, "learning_rate": 9.413937424292791e-05, "loss": 0.0295359306037426, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.02998, "step": 106, "tokens/total": 3208320, "tokens/train_per_sec_per_gpu": 34.18, "tokens/trainable": 48447} +{"epoch": 0.41796875, "grad_norm": 0.481608510017395, "learning_rate": 9.399531154980424e-05, "loss": 0.017734361812472343, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.01789, "step": 107, "tokens/total": 3238416, "tokens/train_per_sec_per_gpu": 30.78, "tokens/trainable": 48889} +{"epoch": 0.421875, "grad_norm": 0.3913033604621887, "learning_rate": 9.384962610259455e-05, "loss": 0.022145511582493782, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.02239, "step": 108, "tokens/total": 3268784, "tokens/train_per_sec_per_gpu": 38.03, "tokens/trainable": 49360} +{"epoch": 0.42578125, "grad_norm": 0.1751519739627838, "learning_rate": 9.370232396386494e-05, "loss": 0.0074960459023714066, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00752, "step": 109, "tokens/total": 3298912, "tokens/train_per_sec_per_gpu": 36.89, "tokens/trainable": 49827} +{"epoch": 0.4296875, "grad_norm": 0.28872016072273254, "learning_rate": 9.355341126345868e-05, "loss": 0.01847866177558899, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.01865, "step": 110, "tokens/total": 3329232, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 50267} +{"epoch": 0.43359375, "grad_norm": 0.3326593041419983, "learning_rate": 9.340289419824107e-05, "loss": 0.013967925682663918, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01407, "step": 111, "tokens/total": 3359440, "tokens/train_per_sec_per_gpu": 33.47, "tokens/trainable": 50711} +{"epoch": 0.4375, "grad_norm": 0.47660934925079346, "learning_rate": 9.325077903184159e-05, "loss": 0.02260858193039894, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.02287, "step": 112, "tokens/total": 3389472, "tokens/train_per_sec_per_gpu": 31.72, "tokens/trainable": 51157} +{"epoch": 0.44140625, "grad_norm": 0.3481523096561432, "learning_rate": 9.30970720943932e-05, "loss": 0.008511725813150406, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00855, "step": 113, "tokens/total": 3419920, "tokens/train_per_sec_per_gpu": 32.79, "tokens/trainable": 51613} +{"epoch": 0.4453125, "grad_norm": 0.4288201928138733, "learning_rate": 9.2941779782269e-05, "loss": 0.02450375072658062, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.02481, "step": 114, "tokens/total": 3450176, "tokens/train_per_sec_per_gpu": 29.19, "tokens/trainable": 52017} +{"epoch": 0.44921875, "grad_norm": 0.17284883558750153, "learning_rate": 9.278490855781596e-05, "loss": 0.007882107980549335, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00791, "step": 115, "tokens/total": 3480544, "tokens/train_per_sec_per_gpu": 36.36, "tokens/trainable": 52506} +{"epoch": 0.453125, "grad_norm": 0.2693459689617157, "learning_rate": 9.262646494908604e-05, "loss": 0.005466018337756395, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00548, "step": 116, "tokens/total": 3510848, "tokens/train_per_sec_per_gpu": 38.77, "tokens/trainable": 53008} +{"epoch": 0.45703125, "grad_norm": 0.48690056800842285, "learning_rate": 9.246645554956457e-05, "loss": 0.004998547025024891, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00501, "step": 117, "tokens/total": 3541296, "tokens/train_per_sec_per_gpu": 35.47, "tokens/trainable": 53467} +{"epoch": 0.4609375, "grad_norm": 0.43558549880981445, "learning_rate": 9.230488701789578e-05, "loss": 0.017219291999936104, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.01737, "step": 118, "tokens/total": 3571936, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 53923} +{"epoch": 0.46484375, "grad_norm": 0.28566548228263855, "learning_rate": 9.214176607760577e-05, "loss": 0.00544874370098114, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00546, "step": 119, "tokens/total": 3602000, "tokens/train_per_sec_per_gpu": 35.72, "tokens/trainable": 54363} +{"epoch": 0.46875, "grad_norm": 0.07611701637506485, "learning_rate": 9.197709951682268e-05, "loss": 0.0014362478395923972, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00144, "step": 120, "tokens/total": 3632288, "tokens/train_per_sec_per_gpu": 35.29, "tokens/trainable": 54839} +{"epoch": 0.47265625, "grad_norm": 0.13201506435871124, "learning_rate": 9.181089418799428e-05, "loss": 0.0024022101424634457, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00241, "step": 121, "tokens/total": 3662608, "tokens/train_per_sec_per_gpu": 32.7, "tokens/trainable": 55267} +{"epoch": 0.4765625, "grad_norm": 0.40975725650787354, "learning_rate": 9.164315700760271e-05, "loss": 0.008171994239091873, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00821, "step": 122, "tokens/total": 3692864, "tokens/train_per_sec_per_gpu": 30.6, "tokens/trainable": 55700} +{"epoch": 0.48046875, "grad_norm": 0.15150536596775055, "learning_rate": 9.147389495587671e-05, "loss": 0.0036906469613313675, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.0037, "step": 123, "tokens/total": 3722864, "tokens/train_per_sec_per_gpu": 33.48, "tokens/trainable": 56144} +{"epoch": 0.484375, "grad_norm": 0.514573335647583, "learning_rate": 9.130311507650116e-05, "loss": 0.01570757105946541, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.01583, "step": 124, "tokens/total": 3753056, "tokens/train_per_sec_per_gpu": 33.18, "tokens/trainable": 56589} +{"epoch": 0.48828125, "grad_norm": 0.43223315477371216, "learning_rate": 9.113082447632394e-05, "loss": 0.007733152247965336, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00776, "step": 125, "tokens/total": 3783312, "tokens/train_per_sec_per_gpu": 36.78, "tokens/trainable": 57081} +{"epoch": 0.4921875, "grad_norm": 0.2972177565097809, "learning_rate": 9.09570303250602e-05, "loss": 0.008276824839413166, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00831, "step": 126, "tokens/total": 3813712, "tokens/train_per_sec_per_gpu": 32.97, "tokens/trainable": 57527} +{"epoch": 0.49609375, "grad_norm": 0.6973468065261841, "learning_rate": 9.078173985499394e-05, "loss": 0.011385848745703697, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01145, "step": 127, "tokens/total": 3844336, "tokens/train_per_sec_per_gpu": 34.98, "tokens/trainable": 58005} +{"epoch": 0.5, "grad_norm": 0.11440204083919525, "learning_rate": 9.060496036067713e-05, "loss": 0.0026649129576981068, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.00267, "step": 128, "tokens/total": 3872688, "tokens/train_per_sec_per_gpu": 35.34, "tokens/trainable": 58454} +{"epoch": 0.50390625, "grad_norm": 0.4737952649593353, "learning_rate": 9.042669919862615e-05, "loss": 0.011923262849450111, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.64, "memory/max_allocated (GiB)": 33.64, "ppl": 1.01199, "step": 129, "tokens/total": 3902640, "tokens/train_per_sec_per_gpu": 32.11, "tokens/trainable": 58896} +{"epoch": 0.5078125, "grad_norm": 0.22904613614082336, "learning_rate": 9.024696378701557e-05, "loss": 0.003364444011822343, "memory/device_reserved (GiB)": 34.68, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00337, "step": 130, "tokens/total": 3933104, "tokens/train_per_sec_per_gpu": 35.96, "tokens/trainable": 59356} +{"epoch": 0.51171875, "grad_norm": 0.6996368169784546, "learning_rate": 9.006576160536948e-05, "loss": 0.0145557951182127, "memory/device_reserved (GiB)": 34.92, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.01466, "step": 131, "tokens/total": 3963184, "tokens/train_per_sec_per_gpu": 32.33, "tokens/trainable": 59792} +{"epoch": 0.515625, "grad_norm": 0.18643876910209656, "learning_rate": 8.988310019425035e-05, "loss": 0.004168116021901369, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00418, "step": 132, "tokens/total": 3993648, "tokens/train_per_sec_per_gpu": 35.79, "tokens/trainable": 60266} +{"epoch": 0.51953125, "grad_norm": 0.3754265606403351, "learning_rate": 8.969898715494506e-05, "loss": 0.01239765528589487, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01247, "step": 133, "tokens/total": 4024000, "tokens/train_per_sec_per_gpu": 35.46, "tokens/trainable": 60759} +{"epoch": 0.5234375, "grad_norm": 0.5720171332359314, "learning_rate": 8.951343014914869e-05, "loss": 0.028643302619457245, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.02906, "step": 134, "tokens/total": 4052480, "tokens/train_per_sec_per_gpu": 31.88, "tokens/trainable": 61204} +{"epoch": 0.52734375, "grad_norm": 0.21384330093860626, "learning_rate": 8.932643689864568e-05, "loss": 0.005144410766661167, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00516, "step": 135, "tokens/total": 4082864, "tokens/train_per_sec_per_gpu": 36.1, "tokens/trainable": 61684} +{"epoch": 0.53125, "grad_norm": 0.3025348484516144, "learning_rate": 8.913801518498845e-05, "loss": 0.011437858454883099, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.7, "memory/max_allocated (GiB)": 33.7, "ppl": 1.0115, "step": 136, "tokens/total": 4113040, "tokens/train_per_sec_per_gpu": 38.36, "tokens/trainable": 62159} +{"epoch": 0.53515625, "grad_norm": 0.16174718737602234, "learning_rate": 8.894817284917364e-05, "loss": 0.005412050988525152, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00543, "step": 137, "tokens/total": 4143344, "tokens/train_per_sec_per_gpu": 35.39, "tokens/trainable": 62649} +{"epoch": 0.5390625, "grad_norm": 0.10317311435937881, "learning_rate": 8.875691779131569e-05, "loss": 0.002703815931454301, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00271, "step": 138, "tokens/total": 4173648, "tokens/train_per_sec_per_gpu": 36.72, "tokens/trainable": 63100} +{"epoch": 0.54296875, "grad_norm": 0.23210155963897705, "learning_rate": 8.856425797031829e-05, "loss": 0.00892355665564537, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00896, "step": 139, "tokens/total": 4204080, "tokens/train_per_sec_per_gpu": 37.69, "tokens/trainable": 63588} +{"epoch": 0.546875, "grad_norm": 0.4230027496814728, "learning_rate": 8.837020140354295e-05, "loss": 0.026184625923633575, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.42, "memory/max_allocated (GiB)": 33.42, "ppl": 1.02653, "step": 140, "tokens/total": 4232400, "tokens/train_per_sec_per_gpu": 33.19, "tokens/trainable": 64005} +{"epoch": 0.55078125, "grad_norm": 0.5172422528266907, "learning_rate": 8.817475616647554e-05, "loss": 0.004686606116592884, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0047, "step": 141, "tokens/total": 4262624, "tokens/train_per_sec_per_gpu": 35.22, "tokens/trainable": 64462} +{"epoch": 0.5546875, "grad_norm": 0.28759023547172546, "learning_rate": 8.797793039239017e-05, "loss": 0.009430551901459694, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00948, "step": 142, "tokens/total": 4293168, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 64900} +{"epoch": 0.55859375, "grad_norm": 0.45416516065597534, "learning_rate": 8.777973227201069e-05, "loss": 0.025679660961031914, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.02601, "step": 143, "tokens/total": 4323248, "tokens/train_per_sec_per_gpu": 37.66, "tokens/trainable": 65379} +{"epoch": 0.5625, "grad_norm": 0.09341549873352051, "learning_rate": 8.758017005316988e-05, "loss": 0.0024609356187283993, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00246, "step": 144, "tokens/total": 4353344, "tokens/train_per_sec_per_gpu": 36.79, "tokens/trainable": 65845} +{"epoch": 0.56640625, "grad_norm": 0.06811556965112686, "learning_rate": 8.737925204046629e-05, "loss": 0.00176865397952497, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00177, "step": 145, "tokens/total": 4383584, "tokens/train_per_sec_per_gpu": 33.07, "tokens/trainable": 66285} +{"epoch": 0.5703125, "grad_norm": 0.3615569472312927, "learning_rate": 8.717698659491851e-05, "loss": 0.018001921474933624, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.01816, "step": 146, "tokens/total": 4413856, "tokens/train_per_sec_per_gpu": 31.66, "tokens/trainable": 66736} +{"epoch": 0.57421875, "grad_norm": 0.37582314014434814, "learning_rate": 8.697338213361735e-05, "loss": 0.01264512725174427, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01273, "step": 147, "tokens/total": 4444112, "tokens/train_per_sec_per_gpu": 35.33, "tokens/trainable": 67150} +{"epoch": 0.578125, "grad_norm": 0.3228939473628998, "learning_rate": 8.676844712937552e-05, "loss": 0.009639455936849117, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00969, "step": 148, "tokens/total": 4474464, "tokens/train_per_sec_per_gpu": 35.2, "tokens/trainable": 67633} +{"epoch": 0.58203125, "grad_norm": 0.44322797656059265, "learning_rate": 8.656219011037509e-05, "loss": 0.013246036134660244, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01333, "step": 149, "tokens/total": 4504880, "tokens/train_per_sec_per_gpu": 33.95, "tokens/trainable": 68077} +{"epoch": 0.5859375, "grad_norm": 0.27842915058135986, "learning_rate": 8.63546196598125e-05, "loss": 0.007822944782674313, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00785, "step": 150, "tokens/total": 4534976, "tokens/train_per_sec_per_gpu": 33.33, "tokens/trainable": 68545} +{"epoch": 0.58984375, "grad_norm": 0.20541682839393616, "learning_rate": 8.614574441554145e-05, "loss": 0.005034038331359625, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00505, "step": 151, "tokens/total": 4565088, "tokens/train_per_sec_per_gpu": 30.33, "tokens/trainable": 69011} +{"epoch": 0.59375, "grad_norm": 0.33200570940971375, "learning_rate": 8.593557306971349e-05, "loss": 0.007599984761327505, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00763, "step": 152, "tokens/total": 4595600, "tokens/train_per_sec_per_gpu": 35.5, "tokens/trainable": 69465} +{"epoch": 0.59765625, "grad_norm": 0.47942468523979187, "learning_rate": 8.572411436841618e-05, "loss": 0.011957819573581219, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01203, "step": 153, "tokens/total": 4626032, "tokens/train_per_sec_per_gpu": 34.03, "tokens/trainable": 69928} +{"epoch": 0.6015625, "grad_norm": 0.13327482342720032, "learning_rate": 8.551137711130922e-05, "loss": 0.0018434002995491028, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00185, "step": 154, "tokens/total": 4656352, "tokens/train_per_sec_per_gpu": 36.8, "tokens/trainable": 70395} +{"epoch": 0.60546875, "grad_norm": 0.2294115424156189, "learning_rate": 8.529737015125824e-05, "loss": 0.003923698328435421, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00393, "step": 155, "tokens/total": 4684672, "tokens/train_per_sec_per_gpu": 35.69, "tokens/trainable": 70829} +{"epoch": 0.609375, "grad_norm": 0.18192212283611298, "learning_rate": 8.508210239396639e-05, "loss": 0.0024503993336111307, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00245, "step": 156, "tokens/total": 4715216, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 71287} +{"epoch": 0.61328125, "grad_norm": 0.015309900976717472, "learning_rate": 8.486558279760375e-05, "loss": 0.0003224749234504998, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00032, "step": 157, "tokens/total": 4745376, "tokens/train_per_sec_per_gpu": 35.1, "tokens/trainable": 71749} +{"epoch": 0.6171875, "grad_norm": 0.39558145403862, "learning_rate": 8.464782037243449e-05, "loss": 0.02732035145163536, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.0277, "step": 158, "tokens/total": 4775776, "tokens/train_per_sec_per_gpu": 32.33, "tokens/trainable": 72185} +{"epoch": 0.62109375, "grad_norm": 0.2731363773345947, "learning_rate": 8.442882418044202e-05, "loss": 0.00566218001767993, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00568, "step": 159, "tokens/total": 4805856, "tokens/train_per_sec_per_gpu": 30.68, "tokens/trainable": 72604} +{"epoch": 0.625, "grad_norm": 0.41349318623542786, "learning_rate": 8.420860333495179e-05, "loss": 0.00760249700397253, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00763, "step": 160, "tokens/total": 4836304, "tokens/train_per_sec_per_gpu": 31.65, "tokens/trainable": 73061} +{"epoch": 0.62890625, "grad_norm": 0.1329299956560135, "learning_rate": 8.398716700025208e-05, "loss": 0.002723761135712266, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00273, "step": 161, "tokens/total": 4866544, "tokens/train_per_sec_per_gpu": 30.1, "tokens/trainable": 73474} +{"epoch": 0.6328125, "grad_norm": 0.2733931839466095, "learning_rate": 8.376452439121266e-05, "loss": 0.008426842279732227, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00846, "step": 162, "tokens/total": 4897120, "tokens/train_per_sec_per_gpu": 32.76, "tokens/trainable": 73923} +{"epoch": 0.63671875, "grad_norm": 0.6801484227180481, "learning_rate": 8.354068477290124e-05, "loss": 0.02508145570755005, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0254, "step": 163, "tokens/total": 4927440, "tokens/train_per_sec_per_gpu": 35.46, "tokens/trainable": 74379} +{"epoch": 0.640625, "grad_norm": 0.22900699079036713, "learning_rate": 8.331565746019807e-05, "loss": 0.006623784080147743, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00665, "step": 164, "tokens/total": 4958032, "tokens/train_per_sec_per_gpu": 33.52, "tokens/trainable": 74830} +{"epoch": 0.64453125, "grad_norm": 0.1291102021932602, "learning_rate": 8.308945181740812e-05, "loss": 0.0024572773836553097, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00246, "step": 165, "tokens/total": 4988368, "tokens/train_per_sec_per_gpu": 31.84, "tokens/trainable": 75262} +{"epoch": 0.6484375, "grad_norm": 0.12582950294017792, "learning_rate": 8.286207725787153e-05, "loss": 0.0014331504935398698, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00143, "step": 166, "tokens/total": 5018672, "tokens/train_per_sec_per_gpu": 31.93, "tokens/trainable": 75703} +{"epoch": 0.65234375, "grad_norm": 0.10334278643131256, "learning_rate": 8.263354324357182e-05, "loss": 0.003322131698951125, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00333, "step": 167, "tokens/total": 5049344, "tokens/train_per_sec_per_gpu": 31.36, "tokens/trainable": 76164} +{"epoch": 0.65625, "grad_norm": 0.344831258058548, "learning_rate": 8.240385928474219e-05, "loss": 0.01979166269302368, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.01999, "step": 168, "tokens/total": 5079856, "tokens/train_per_sec_per_gpu": 34.58, "tokens/trainable": 76623} +{"epoch": 0.66015625, "grad_norm": 0.3123904764652252, "learning_rate": 8.217303493946967e-05, "loss": 0.020188894122838974, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.02039, "step": 169, "tokens/total": 5110192, "tokens/train_per_sec_per_gpu": 34.46, "tokens/trainable": 77047} +{"epoch": 0.6640625, "grad_norm": 0.24240995943546295, "learning_rate": 8.194107981329746e-05, "loss": 0.01431284286081791, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01442, "step": 170, "tokens/total": 5140592, "tokens/train_per_sec_per_gpu": 34.17, "tokens/trainable": 77496} +{"epoch": 0.66796875, "grad_norm": 0.25977423787117004, "learning_rate": 8.170800355882518e-05, "loss": 0.014354058541357517, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.01446, "step": 171, "tokens/total": 5170848, "tokens/train_per_sec_per_gpu": 33.29, "tokens/trainable": 77956} +{"epoch": 0.671875, "grad_norm": 0.19354988634586334, "learning_rate": 8.147381587530713e-05, "loss": 0.004821081180125475, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00483, "step": 172, "tokens/total": 5201280, "tokens/train_per_sec_per_gpu": 29.87, "tokens/trainable": 78410} +{"epoch": 0.67578125, "grad_norm": 0.2551465928554535, "learning_rate": 8.123852650824877e-05, "loss": 0.010946989059448242, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.01101, "step": 173, "tokens/total": 5231712, "tokens/train_per_sec_per_gpu": 33.2, "tokens/trainable": 78849} +{"epoch": 0.6796875, "grad_norm": 0.34295886754989624, "learning_rate": 8.100214524900103e-05, "loss": 0.02883588895201683, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.62, "memory/max_allocated (GiB)": 33.62, "ppl": 1.02926, "step": 174, "tokens/total": 5261440, "tokens/train_per_sec_per_gpu": 30.2, "tokens/trainable": 79266} +{"epoch": 0.68359375, "grad_norm": 0.0546368770301342, "learning_rate": 8.076468193435301e-05, "loss": 0.0018831162014976144, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00188, "step": 175, "tokens/total": 5291728, "tokens/train_per_sec_per_gpu": 32.58, "tokens/trainable": 79736} +{"epoch": 0.6875, "grad_norm": 0.22756348550319672, "learning_rate": 8.052614644612253e-05, "loss": 0.0047474936582148075, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00476, "step": 176, "tokens/total": 5322240, "tokens/train_per_sec_per_gpu": 36.29, "tokens/trainable": 80214} +{"epoch": 0.69140625, "grad_norm": 0.06755488365888596, "learning_rate": 8.028654871074489e-05, "loss": 0.0017190770013257861, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00172, "step": 177, "tokens/total": 5352576, "tokens/train_per_sec_per_gpu": 32.58, "tokens/trainable": 80661} +{"epoch": 0.6953125, "grad_norm": 0.2727944850921631, "learning_rate": 8.004589869885986e-05, "loss": 0.00690429238602519, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00693, "step": 178, "tokens/total": 5382784, "tokens/train_per_sec_per_gpu": 34.86, "tokens/trainable": 81129} +{"epoch": 0.69921875, "grad_norm": 0.20468388497829437, "learning_rate": 7.980420642489674e-05, "loss": 0.003946482669562101, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00395, "step": 179, "tokens/total": 5412784, "tokens/train_per_sec_per_gpu": 33.07, "tokens/trainable": 81576} +{"epoch": 0.703125, "grad_norm": 0.25605547428131104, "learning_rate": 7.95614819466576e-05, "loss": 0.004923749715089798, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00494, "step": 180, "tokens/total": 5442960, "tokens/train_per_sec_per_gpu": 34.63, "tokens/trainable": 82049} +{"epoch": 0.70703125, "grad_norm": 0.33550992608070374, "learning_rate": 7.931773536489872e-05, "loss": 0.010120240971446037, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01017, "step": 181, "tokens/total": 5473216, "tokens/train_per_sec_per_gpu": 38.84, "tokens/trainable": 82566} +{"epoch": 0.7109375, "grad_norm": 0.22376962006092072, "learning_rate": 7.907297682291035e-05, "loss": 0.0025828760117292404, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00259, "step": 182, "tokens/total": 5503568, "tokens/train_per_sec_per_gpu": 33.98, "tokens/trainable": 83054} +{"epoch": 0.71484375, "grad_norm": 0.023590704426169395, "learning_rate": 7.882721650609442e-05, "loss": 0.0007615481736138463, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00076, "step": 183, "tokens/total": 5533760, "tokens/train_per_sec_per_gpu": 32.84, "tokens/trainable": 83501} +{"epoch": 0.71875, "grad_norm": 0.06294015794992447, "learning_rate": 7.85804646415409e-05, "loss": 0.0010393422562628984, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00104, "step": 184, "tokens/total": 5564240, "tokens/train_per_sec_per_gpu": 35.58, "tokens/trainable": 84007} +{"epoch": 0.72265625, "grad_norm": 0.04545356705784798, "learning_rate": 7.833273149760207e-05, "loss": 0.0006121923215687275, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00061, "step": 185, "tokens/total": 5594448, "tokens/train_per_sec_per_gpu": 35.61, "tokens/trainable": 84491} +{"epoch": 0.7265625, "grad_norm": 0.23753786087036133, "learning_rate": 7.808402738346527e-05, "loss": 0.00197276147082448, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00197, "step": 186, "tokens/total": 5624816, "tokens/train_per_sec_per_gpu": 33.44, "tokens/trainable": 84952} +{"epoch": 0.73046875, "grad_norm": 0.12784555554389954, "learning_rate": 7.783436264872382e-05, "loss": 0.0010586264543235302, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00106, "step": 187, "tokens/total": 5655104, "tokens/train_per_sec_per_gpu": 33.02, "tokens/trainable": 85400} +{"epoch": 0.734375, "grad_norm": 0.8053486347198486, "learning_rate": 7.758374768294647e-05, "loss": 0.01503228023648262, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01515, "step": 188, "tokens/total": 5685360, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 85889} +{"epoch": 0.73828125, "grad_norm": 0.6335012316703796, "learning_rate": 7.733219291524489e-05, "loss": 0.004618915729224682, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00463, "step": 189, "tokens/total": 5715648, "tokens/train_per_sec_per_gpu": 36.31, "tokens/trainable": 86396} +{"epoch": 0.7421875, "grad_norm": 0.7543070912361145, "learning_rate": 7.707970881383977e-05, "loss": 0.03686961531639099, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.03756, "step": 190, "tokens/total": 5746240, "tokens/train_per_sec_per_gpu": 33.92, "tokens/trainable": 86867} +{"epoch": 0.74609375, "grad_norm": 0.5514787435531616, "learning_rate": 7.682630588562518e-05, "loss": 0.023130130022764206, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.0234, "step": 191, "tokens/total": 5776816, "tokens/train_per_sec_per_gpu": 36.24, "tokens/trainable": 87314} +{"epoch": 0.75, "grad_norm": 0.30548200011253357, "learning_rate": 7.657199467573129e-05, "loss": 0.002730845008045435, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00273, "step": 192, "tokens/total": 5807248, "tokens/train_per_sec_per_gpu": 30.41, "tokens/trainable": 87736} +{"epoch": 0.75390625, "grad_norm": 0.42851904034614563, "learning_rate": 7.631678576708561e-05, "loss": 0.015687599778175354, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01581, "step": 193, "tokens/total": 5837680, "tokens/train_per_sec_per_gpu": 35.15, "tokens/trainable": 88219} +{"epoch": 0.7578125, "grad_norm": 0.14844387769699097, "learning_rate": 7.606068977997255e-05, "loss": 0.0037631455343216658, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00377, "step": 194, "tokens/total": 5868000, "tokens/train_per_sec_per_gpu": 36.26, "tokens/trainable": 88699} +{"epoch": 0.76171875, "grad_norm": 0.19134196639060974, "learning_rate": 7.580371737159148e-05, "loss": 0.004276394844055176, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.38, "memory/max_allocated (GiB)": 33.38, "ppl": 1.00429, "step": 195, "tokens/total": 5896224, "tokens/train_per_sec_per_gpu": 30.51, "tokens/trainable": 89132} +{"epoch": 0.765625, "grad_norm": 0.44769415259361267, "learning_rate": 7.554587923561324e-05, "loss": 0.013675286434590816, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01377, "step": 196, "tokens/total": 5926624, "tokens/train_per_sec_per_gpu": 37.36, "tokens/trainable": 89594} +{"epoch": 0.76953125, "grad_norm": 0.11093451082706451, "learning_rate": 7.528718610173511e-05, "loss": 0.0035221045836806297, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.34, "memory/max_allocated (GiB)": 33.34, "ppl": 1.00353, "step": 197, "tokens/total": 5954944, "tokens/train_per_sec_per_gpu": 38.22, "tokens/trainable": 90042} +{"epoch": 0.7734375, "grad_norm": 0.19566713273525238, "learning_rate": 7.502764873523431e-05, "loss": 0.00824933685362339, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00828, "step": 198, "tokens/total": 5985200, "tokens/train_per_sec_per_gpu": 34.23, "tokens/trainable": 90503} +{"epoch": 0.77734375, "grad_norm": 0.1111554354429245, "learning_rate": 7.476727793652011e-05, "loss": 0.002672867150977254, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 34.0, "memory/max_allocated (GiB)": 34.0, "ppl": 1.00268, "step": 199, "tokens/total": 6015952, "tokens/train_per_sec_per_gpu": 36.63, "tokens/trainable": 90982} +{"epoch": 0.78125, "grad_norm": 0.2508006989955902, "learning_rate": 7.450608454068415e-05, "loss": 0.0060717021115124226, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00609, "step": 200, "tokens/total": 6046208, "tokens/train_per_sec_per_gpu": 39.5, "tokens/trainable": 91466} +{"epoch": 0.78515625, "grad_norm": 0.07083698362112045, "learning_rate": 7.424407941704987e-05, "loss": 0.0025744480080902576, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00258, "step": 201, "tokens/total": 6076208, "tokens/train_per_sec_per_gpu": 34.38, "tokens/trainable": 91912} +{"epoch": 0.7890625, "grad_norm": 0.06263043731451035, "learning_rate": 7.398127346871986e-05, "loss": 0.00145058985799551, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00145, "step": 202, "tokens/total": 6106560, "tokens/train_per_sec_per_gpu": 37.95, "tokens/trainable": 92385} +{"epoch": 0.79296875, "grad_norm": 0.44214949011802673, "learning_rate": 7.371767763212238e-05, "loss": 0.02974056825041771, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.03019, "step": 203, "tokens/total": 6136640, "tokens/train_per_sec_per_gpu": 36.69, "tokens/trainable": 92842} +{"epoch": 0.796875, "grad_norm": 0.39396584033966064, "learning_rate": 7.345330287655617e-05, "loss": 0.011925775557756424, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.012, "step": 204, "tokens/total": 6167184, "tokens/train_per_sec_per_gpu": 34.1, "tokens/trainable": 93323} +{"epoch": 0.80078125, "grad_norm": 0.5901291370391846, "learning_rate": 7.31881602037339e-05, "loss": 0.019959867000579834, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.02016, "step": 205, "tokens/total": 6197376, "tokens/train_per_sec_per_gpu": 34.91, "tokens/trainable": 93756} +{"epoch": 0.8046875, "grad_norm": 0.2788645327091217, "learning_rate": 7.29222606473245e-05, "loss": 0.006857238709926605, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00688, "step": 206, "tokens/total": 6227600, "tokens/train_per_sec_per_gpu": 37.66, "tokens/trainable": 94235} +{"epoch": 0.80859375, "grad_norm": 0.19782070815563202, "learning_rate": 7.265561527249383e-05, "loss": 0.005047307349741459, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00506, "step": 207, "tokens/total": 6257760, "tokens/train_per_sec_per_gpu": 36.62, "tokens/trainable": 94733} +{"epoch": 0.8125, "grad_norm": 0.14907793700695038, "learning_rate": 7.238823517544436e-05, "loss": 0.0029104873538017273, "memory/device_reserved (GiB)": 35.93, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00291, "step": 208, "tokens/total": 6288048, "tokens/train_per_sec_per_gpu": 33.9, "tokens/trainable": 95218} +{"epoch": 0.81640625, "grad_norm": 0.03554432839155197, "learning_rate": 7.212013148295333e-05, "loss": 0.0010063875233754516, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00101, "step": 209, "tokens/total": 6318688, "tokens/train_per_sec_per_gpu": 29.31, "tokens/trainable": 95619} +{"epoch": 0.8203125, "grad_norm": 0.11122506856918335, "learning_rate": 7.185131535190975e-05, "loss": 0.0032706214115023613, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00328, "step": 210, "tokens/total": 6348944, "tokens/train_per_sec_per_gpu": 35.51, "tokens/trainable": 96076} +{"epoch": 0.82421875, "grad_norm": 0.38507047295570374, "learning_rate": 7.158179796885005e-05, "loss": 0.006026511080563068, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00604, "step": 211, "tokens/total": 6378960, "tokens/train_per_sec_per_gpu": 35.43, "tokens/trainable": 96505} +{"epoch": 0.828125, "grad_norm": 0.5194666385650635, "learning_rate": 7.131159054949273e-05, "loss": 0.008640932850539684, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00868, "step": 212, "tokens/total": 6409472, "tokens/train_per_sec_per_gpu": 32.76, "tokens/trainable": 96930} +{"epoch": 0.83203125, "grad_norm": 0.0659736841917038, "learning_rate": 7.104070433827139e-05, "loss": 0.0014961843844503164, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0015, "step": 213, "tokens/total": 6439904, "tokens/train_per_sec_per_gpu": 32.82, "tokens/trainable": 97361} +{"epoch": 0.8359375, "grad_norm": 0.233027845621109, "learning_rate": 7.076915060786705e-05, "loss": 0.00338245602324605, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00339, "step": 214, "tokens/total": 6470176, "tokens/train_per_sec_per_gpu": 36.41, "tokens/trainable": 97858} +{"epoch": 0.83984375, "grad_norm": 0.38389813899993896, "learning_rate": 7.049694065873882e-05, "loss": 0.014850301668047905, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01496, "step": 215, "tokens/total": 6500336, "tokens/train_per_sec_per_gpu": 37.61, "tokens/trainable": 98308} +{"epoch": 0.84375, "grad_norm": 0.07625724375247955, "learning_rate": 7.022408581865382e-05, "loss": 0.0009395676897838712, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00094, "step": 216, "tokens/total": 6530688, "tokens/train_per_sec_per_gpu": 36.55, "tokens/trainable": 98781} +{"epoch": 0.84765625, "grad_norm": 0.1813930720090866, "learning_rate": 6.99505974422157e-05, "loss": 0.005376772489398718, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00539, "step": 217, "tokens/total": 6561040, "tokens/train_per_sec_per_gpu": 36.83, "tokens/trainable": 99266} +{"epoch": 0.8515625, "grad_norm": 0.28488755226135254, "learning_rate": 6.967648691039213e-05, "loss": 0.007393942214548588, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00742, "step": 218, "tokens/total": 6589392, "tokens/train_per_sec_per_gpu": 39.74, "tokens/trainable": 99731} +{"epoch": 0.85546875, "grad_norm": 0.2636983394622803, "learning_rate": 6.940176563004123e-05, "loss": 0.0031264584977179766, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00313, "step": 219, "tokens/total": 6619824, "tokens/train_per_sec_per_gpu": 31.81, "tokens/trainable": 100164} +{"epoch": 0.859375, "grad_norm": 0.19989553093910217, "learning_rate": 6.912644503343682e-05, "loss": 0.018466733396053314, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01864, "step": 220, "tokens/total": 6650224, "tokens/train_per_sec_per_gpu": 36.33, "tokens/trainable": 100620} +{"epoch": 0.86328125, "grad_norm": 0.1063912957906723, "learning_rate": 6.885053657779273e-05, "loss": 0.0018869235645979643, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00189, "step": 221, "tokens/total": 6680384, "tokens/train_per_sec_per_gpu": 33.02, "tokens/trainable": 101080} +{"epoch": 0.8671875, "grad_norm": 0.41694170236587524, "learning_rate": 6.857405174478604e-05, "loss": 0.025706909596920013, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.02604, "step": 222, "tokens/total": 6710896, "tokens/train_per_sec_per_gpu": 37.93, "tokens/trainable": 101546} +{"epoch": 0.87109375, "grad_norm": 0.7918646335601807, "learning_rate": 6.82970020400792e-05, "loss": 0.01185107696801424, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01192, "step": 223, "tokens/total": 6741280, "tokens/train_per_sec_per_gpu": 31.49, "tokens/trainable": 101987} +{"epoch": 0.875, "grad_norm": 0.32625025510787964, "learning_rate": 6.801939899284132e-05, "loss": 0.009739323519170284, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00979, "step": 224, "tokens/total": 6771488, "tokens/train_per_sec_per_gpu": 35.55, "tokens/trainable": 102435} +{"epoch": 0.87890625, "grad_norm": 0.4342412054538727, "learning_rate": 6.774125415526827e-05, "loss": 0.009223651140928268, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00927, "step": 225, "tokens/total": 6801744, "tokens/train_per_sec_per_gpu": 29.28, "tokens/trainable": 102843} +{"epoch": 0.8828125, "grad_norm": 0.13243624567985535, "learning_rate": 6.746257910210214e-05, "loss": 0.0032694521360099316, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00327, "step": 226, "tokens/total": 6832432, "tokens/train_per_sec_per_gpu": 37.49, "tokens/trainable": 103342} +{"epoch": 0.88671875, "grad_norm": 0.09638779610395432, "learning_rate": 6.718338543014937e-05, "loss": 0.0033315662294626236, "memory/device_reserved (GiB)": 35.71, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00334, "step": 227, "tokens/total": 6862448, "tokens/train_per_sec_per_gpu": 40.85, "tokens/trainable": 103844} +{"epoch": 0.890625, "grad_norm": 0.3315463960170746, "learning_rate": 6.69036847577983e-05, "loss": 0.020162900909781456, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.02037, "step": 228, "tokens/total": 6892832, "tokens/train_per_sec_per_gpu": 36.36, "tokens/trainable": 104307} +{"epoch": 0.89453125, "grad_norm": 0.23732618987560272, "learning_rate": 6.662348872453553e-05, "loss": 0.014629160054028034, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.01474, "step": 229, "tokens/total": 6922832, "tokens/train_per_sec_per_gpu": 33.58, "tokens/trainable": 104772} +{"epoch": 0.8984375, "grad_norm": 0.06660830229520798, "learning_rate": 6.63428089904618e-05, "loss": 0.0017317197052761912, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00173, "step": 230, "tokens/total": 6953104, "tokens/train_per_sec_per_gpu": 36.91, "tokens/trainable": 105240} +{"epoch": 0.90234375, "grad_norm": 0.4719278812408447, "learning_rate": 6.60616572358065e-05, "loss": 0.00895722210407257, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.009, "step": 231, "tokens/total": 6983440, "tokens/train_per_sec_per_gpu": 33.85, "tokens/trainable": 105703} +{"epoch": 0.90625, "grad_norm": 0.18451477587223053, "learning_rate": 6.578004516044172e-05, "loss": 0.0027692944277077913, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00277, "step": 232, "tokens/total": 7013840, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 106179} +{"epoch": 0.91015625, "grad_norm": 0.07695464044809341, "learning_rate": 6.549798448339548e-05, "loss": 0.002346785506233573, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00235, "step": 233, "tokens/total": 7044240, "tokens/train_per_sec_per_gpu": 38.08, "tokens/trainable": 106683} +{"epoch": 0.9140625, "grad_norm": 0.23940840363502502, "learning_rate": 6.521548694236384e-05, "loss": 0.003448254195973277, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00345, "step": 234, "tokens/total": 7074224, "tokens/train_per_sec_per_gpu": 33.95, "tokens/trainable": 107121} +{"epoch": 0.91796875, "grad_norm": 0.26199451088905334, "learning_rate": 6.493256429322259e-05, "loss": 0.005110536701977253, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00512, "step": 235, "tokens/total": 7104672, "tokens/train_per_sec_per_gpu": 32.84, "tokens/trainable": 107561} +{"epoch": 0.921875, "grad_norm": 0.19866062700748444, "learning_rate": 6.464922830953799e-05, "loss": 0.004088824614882469, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0041, "step": 236, "tokens/total": 7134784, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 108018} +{"epoch": 0.92578125, "grad_norm": 0.024193251505494118, "learning_rate": 6.436549078207688e-05, "loss": 0.0006136820884421468, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00061, "step": 237, "tokens/total": 7164960, "tokens/train_per_sec_per_gpu": 33.45, "tokens/trainable": 108446} +{"epoch": 0.9296875, "grad_norm": 0.05543803051114082, "learning_rate": 6.408136351831592e-05, "loss": 0.0012521358439698815, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00125, "step": 238, "tokens/total": 7195344, "tokens/train_per_sec_per_gpu": 29.3, "tokens/trainable": 108866} +{"epoch": 0.93359375, "grad_norm": 0.7693765163421631, "learning_rate": 6.379685834195036e-05, "loss": 0.010034173727035522, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01008, "step": 239, "tokens/total": 7225680, "tokens/train_per_sec_per_gpu": 33.92, "tokens/trainable": 109316} +{"epoch": 0.9375, "grad_norm": 0.06719055026769638, "learning_rate": 6.351198709240186e-05, "loss": 0.0014942979905754328, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0015, "step": 240, "tokens/total": 7256112, "tokens/train_per_sec_per_gpu": 33.39, "tokens/trainable": 109787} +{"epoch": 0.94140625, "grad_norm": 0.021327359601855278, "learning_rate": 6.32267616243259e-05, "loss": 0.0003477554419077933, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00035, "step": 241, "tokens/total": 7286384, "tokens/train_per_sec_per_gpu": 38.51, "tokens/trainable": 110246} +{"epoch": 0.9453125, "grad_norm": 0.015047608874738216, "learning_rate": 6.294119380711849e-05, "loss": 0.00033737512421794236, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00034, "step": 242, "tokens/total": 7316688, "tokens/train_per_sec_per_gpu": 29.27, "tokens/trainable": 110667} +{"epoch": 0.94921875, "grad_norm": 0.14928486943244934, "learning_rate": 6.265529552442209e-05, "loss": 0.018782632425427437, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01896, "step": 243, "tokens/total": 7346992, "tokens/train_per_sec_per_gpu": 38.3, "tokens/trainable": 111154} +{"epoch": 0.953125, "grad_norm": 0.005887618288397789, "learning_rate": 6.236907867363127e-05, "loss": 0.00017539145483169705, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00018, "step": 244, "tokens/total": 7377632, "tokens/train_per_sec_per_gpu": 33.16, "tokens/trainable": 111611} +{"epoch": 0.95703125, "grad_norm": 0.04487505182623863, "learning_rate": 6.208255516539749e-05, "loss": 0.0005144703900441527, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00051, "step": 245, "tokens/total": 7407872, "tokens/train_per_sec_per_gpu": 34.29, "tokens/trainable": 112102} +{"epoch": 0.9609375, "grad_norm": 0.3277672231197357, "learning_rate": 6.179573692313344e-05, "loss": 0.010591475293040276, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.01065, "step": 246, "tokens/total": 7438448, "tokens/train_per_sec_per_gpu": 33.31, "tokens/trainable": 112569} +{"epoch": 0.96484375, "grad_norm": 0.010867438279092312, "learning_rate": 6.150863588251694e-05, "loss": 0.0003141904016956687, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00031, "step": 247, "tokens/total": 7466768, "tokens/train_per_sec_per_gpu": 37.79, "tokens/trainable": 113023} +{"epoch": 0.96875, "grad_norm": 0.011889747343957424, "learning_rate": 6.122126399099419e-05, "loss": 0.0003558638272807002, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00036, "step": 248, "tokens/total": 7497360, "tokens/train_per_sec_per_gpu": 35.25, "tokens/trainable": 113464} +{"epoch": 0.97265625, "grad_norm": 0.040197841823101044, "learning_rate": 6.0933633207282615e-05, "loss": 0.0003707111463882029, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00037, "step": 249, "tokens/total": 7527504, "tokens/train_per_sec_per_gpu": 32.79, "tokens/trainable": 113908} +{"epoch": 0.9765625, "grad_norm": 0.01977125182747841, "learning_rate": 6.064575550087316e-05, "loss": 0.000563526526093483, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00056, "step": 250, "tokens/total": 7555808, "tokens/train_per_sec_per_gpu": 35.43, "tokens/trainable": 114348} +{"epoch": 0.98046875, "grad_norm": 0.04650595411658287, "learning_rate": 6.0357642851532245e-05, "loss": 0.0004946058033965528, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00049, "step": 251, "tokens/total": 7586320, "tokens/train_per_sec_per_gpu": 35.3, "tokens/trainable": 114830} +{"epoch": 0.984375, "grad_norm": 0.3321515619754791, "learning_rate": 6.0069307248803294e-05, "loss": 0.0047949617728590965, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00481, "step": 252, "tokens/total": 7616576, "tokens/train_per_sec_per_gpu": 35.25, "tokens/trainable": 115278} +{"epoch": 0.98828125, "grad_norm": 0.248526468873024, "learning_rate": 5.9780760691507635e-05, "loss": 0.009970192797482014, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01002, "step": 253, "tokens/total": 7647152, "tokens/train_per_sec_per_gpu": 32.45, "tokens/trainable": 115699} +{"epoch": 0.9921875, "grad_norm": 0.01950794644653797, "learning_rate": 5.9492015187245334e-05, "loss": 0.000543278525583446, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00054, "step": 254, "tokens/total": 7677584, "tokens/train_per_sec_per_gpu": 30.85, "tokens/trainable": 116133} +{"epoch": 0.99609375, "grad_norm": 0.013262342661619186, "learning_rate": 5.920308275189541e-05, "loss": 0.00043126061791554093, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00043, "step": 255, "tokens/total": 7708016, "tokens/train_per_sec_per_gpu": 36.89, "tokens/trainable": 116591} +{"epoch": 1.0, "grad_norm": 0.008060271851718426, "learning_rate": 5.8913975409115874e-05, "loss": 0.0002502937277313322, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00025, "step": 256, "tokens/total": 7738608, "tokens/train_per_sec_per_gpu": 32.19, "tokens/trainable": 117039} +{"epoch": 1.00390625, "grad_norm": 0.006059759296476841, "learning_rate": 5.8624705189843395e-05, "loss": 0.000195491302292794, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0002, "step": 257, "tokens/total": 7768944, "tokens/train_per_sec_per_gpu": 32.63, "tokens/trainable": 117495} +{"epoch": 1.0078125, "grad_norm": 0.026264909654855728, "learning_rate": 5.833528413179249e-05, "loss": 0.00039901715354062617, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0004, "step": 258, "tokens/total": 7799232, "tokens/train_per_sec_per_gpu": 30.58, "tokens/trainable": 117925} +{"epoch": 1.01171875, "grad_norm": 0.035495247691869736, "learning_rate": 5.80457242789548e-05, "loss": 0.0005823379615321755, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00058, "step": 259, "tokens/total": 7829568, "tokens/train_per_sec_per_gpu": 31.69, "tokens/trainable": 118373} +{"epoch": 1.015625, "grad_norm": 0.004000507295131683, "learning_rate": 5.77560376810977e-05, "loss": 0.00013823891640640795, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00014, "step": 260, "tokens/total": 7860000, "tokens/train_per_sec_per_gpu": 31.53, "tokens/trainable": 118816} +{"epoch": 1.01953125, "grad_norm": 0.03404498100280762, "learning_rate": 5.7466236393263005e-05, "loss": 0.0006110825343057513, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00061, "step": 261, "tokens/total": 7890464, "tokens/train_per_sec_per_gpu": 33.97, "tokens/trainable": 119263} +{"epoch": 1.0234375, "grad_norm": 0.018318751826882362, "learning_rate": 5.717633247526522e-05, "loss": 0.0002658100565895438, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00027, "step": 262, "tokens/total": 7920944, "tokens/train_per_sec_per_gpu": 34.88, "tokens/trainable": 119772} +{"epoch": 1.02734375, "grad_norm": 0.21364416182041168, "learning_rate": 5.688633799118971e-05, "loss": 0.002921548904851079, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00293, "step": 263, "tokens/total": 7951296, "tokens/train_per_sec_per_gpu": 33.91, "tokens/trainable": 120269} +{"epoch": 1.03125, "grad_norm": 0.014080969616770744, "learning_rate": 5.659626500889066e-05, "loss": 0.000284558511339128, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00028, "step": 264, "tokens/total": 7981856, "tokens/train_per_sec_per_gpu": 37.2, "tokens/trainable": 120746} +{"epoch": 1.03515625, "grad_norm": 0.018712079152464867, "learning_rate": 5.6306125599488905e-05, "loss": 0.0003485676134005189, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00035, "step": 265, "tokens/total": 8012192, "tokens/train_per_sec_per_gpu": 38.41, "tokens/trainable": 121204} +{"epoch": 1.0390625, "grad_norm": 0.16164301335811615, "learning_rate": 5.601593183686955e-05, "loss": 0.004497133661061525, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00451, "step": 266, "tokens/total": 8042736, "tokens/train_per_sec_per_gpu": 31.36, "tokens/trainable": 121670} +{"epoch": 1.04296875, "grad_norm": 0.006513051688671112, "learning_rate": 5.572569579717961e-05, "loss": 0.00010482803918421268, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.0001, "step": 267, "tokens/total": 8073024, "tokens/train_per_sec_per_gpu": 40.18, "tokens/trainable": 122183} +{"epoch": 1.046875, "grad_norm": 0.008712893351912498, "learning_rate": 5.543542955832538e-05, "loss": 0.0001421938359271735, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00014, "step": 268, "tokens/total": 8103280, "tokens/train_per_sec_per_gpu": 35.31, "tokens/trainable": 122645} +{"epoch": 1.05078125, "grad_norm": 0.006345031317323446, "learning_rate": 5.514514519946986e-05, "loss": 0.00020790408598259091, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00021, "step": 269, "tokens/total": 8133440, "tokens/train_per_sec_per_gpu": 33.17, "tokens/trainable": 123105} +{"epoch": 1.0546875, "grad_norm": 0.003045592224225402, "learning_rate": 5.485485480053015e-05, "loss": 8.933185745263472e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00009, "step": 270, "tokens/total": 8163920, "tokens/train_per_sec_per_gpu": 36.28, "tokens/trainable": 123595} +{"epoch": 1.05859375, "grad_norm": 0.10080692917108536, "learning_rate": 5.4564570441674645e-05, "loss": 0.0011391319567337632, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00114, "step": 271, "tokens/total": 8194032, "tokens/train_per_sec_per_gpu": 37.78, "tokens/trainable": 124077} +{"epoch": 1.0625, "grad_norm": 0.01787402294576168, "learning_rate": 5.42743042028204e-05, "loss": 0.000280556152574718, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00028, "step": 272, "tokens/total": 8224320, "tokens/train_per_sec_per_gpu": 34.72, "tokens/trainable": 124547} +{"epoch": 1.06640625, "grad_norm": 0.007194901816546917, "learning_rate": 5.3984068163130464e-05, "loss": 0.00012255870387889445, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00012, "step": 273, "tokens/total": 8254368, "tokens/train_per_sec_per_gpu": 34.08, "tokens/trainable": 124977} +{"epoch": 1.0703125, "grad_norm": 0.05285244435071945, "learning_rate": 5.369387440051111e-05, "loss": 0.0003489117370918393, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00035, "step": 274, "tokens/total": 8284944, "tokens/train_per_sec_per_gpu": 34.63, "tokens/trainable": 125438} +{"epoch": 1.07421875, "grad_norm": 0.004284605849534273, "learning_rate": 5.340373499110935e-05, "loss": 0.000126057377201505, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00013, "step": 275, "tokens/total": 8315584, "tokens/train_per_sec_per_gpu": 31.58, "tokens/trainable": 125889} +{"epoch": 1.078125, "grad_norm": 0.006148909218609333, "learning_rate": 5.3113662008810304e-05, "loss": 0.00012465259351301938, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00012, "step": 276, "tokens/total": 8345952, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 126363} +{"epoch": 1.08203125, "grad_norm": 0.006904160603880882, "learning_rate": 5.282366752473479e-05, "loss": 0.00013310338545124978, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00013, "step": 277, "tokens/total": 8376416, "tokens/train_per_sec_per_gpu": 34.11, "tokens/trainable": 126828} +{"epoch": 1.0859375, "grad_norm": 0.001829590299166739, "learning_rate": 5.2533763606737005e-05, "loss": 4.333024116931483e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00004, "step": 278, "tokens/total": 8406640, "tokens/train_per_sec_per_gpu": 38.3, "tokens/trainable": 127331} +{"epoch": 1.08984375, "grad_norm": 0.07432714849710464, "learning_rate": 5.224396231890232e-05, "loss": 0.0006949692033231258, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.0007, "step": 279, "tokens/total": 8436784, "tokens/train_per_sec_per_gpu": 34.4, "tokens/trainable": 127798} +{"epoch": 1.09375, "grad_norm": 0.048953574150800705, "learning_rate": 5.195427572104522e-05, "loss": 0.0004470825952012092, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00045, "step": 280, "tokens/total": 8467152, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 128247} +{"epoch": 1.09765625, "grad_norm": 0.0006554989377036691, "learning_rate": 5.166471586820751e-05, "loss": 1.6557103663217276e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00002, "step": 281, "tokens/total": 8497424, "tokens/train_per_sec_per_gpu": 36.26, "tokens/trainable": 128729} +{"epoch": 1.1015625, "grad_norm": 0.01792880706489086, "learning_rate": 5.1375294810156615e-05, "loss": 0.00023528770543634892, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00024, "step": 282, "tokens/total": 8527616, "tokens/train_per_sec_per_gpu": 35.15, "tokens/trainable": 129203} +{"epoch": 1.10546875, "grad_norm": 0.07039739936590195, "learning_rate": 5.1086024590884144e-05, "loss": 0.0005169064970687032, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00052, "step": 283, "tokens/total": 8557984, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 129653} +{"epoch": 1.109375, "grad_norm": 0.25704729557037354, "learning_rate": 5.079691724810461e-05, "loss": 0.001955911749973893, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00196, "step": 284, "tokens/total": 8588384, "tokens/train_per_sec_per_gpu": 32.6, "tokens/trainable": 130100} +{"epoch": 1.11328125, "grad_norm": 0.0025027310475707054, "learning_rate": 5.0507984812754684e-05, "loss": 4.597695806296542e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00005, "step": 285, "tokens/total": 8618768, "tokens/train_per_sec_per_gpu": 30.52, "tokens/trainable": 130543} +{"epoch": 1.1171875, "grad_norm": 0.002928070956841111, "learning_rate": 5.021923930849237e-05, "loss": 6.182275683386251e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00006, "step": 286, "tokens/total": 8648976, "tokens/train_per_sec_per_gpu": 34.39, "tokens/trainable": 131000} +{"epoch": 1.12109375, "grad_norm": 0.0009919465519487858, "learning_rate": 4.99306927511967e-05, "loss": 2.293481884407811e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00002, "step": 287, "tokens/total": 8679152, "tokens/train_per_sec_per_gpu": 34.09, "tokens/trainable": 131449} +{"epoch": 1.125, "grad_norm": 0.0005833669565618038, "learning_rate": 4.964235714846775e-05, "loss": 1.711631557554938e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00002, "step": 288, "tokens/total": 8709504, "tokens/train_per_sec_per_gpu": 36.12, "tokens/trainable": 131923} +{"epoch": 1.12890625, "grad_norm": 0.0012934672413393855, "learning_rate": 4.9354244499126866e-05, "loss": 3.084562922595069e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00003, "step": 289, "tokens/total": 8739680, "tokens/train_per_sec_per_gpu": 33.29, "tokens/trainable": 132397} +{"epoch": 1.1328125, "grad_norm": 0.0005865129642188549, "learning_rate": 4.90663667927174e-05, "loss": 1.5193030776572414e-05, "memory/device_reserved (GiB)": 35.57, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00002, "step": 290, "tokens/total": 8770112, "tokens/train_per_sec_per_gpu": 36.47, "tokens/trainable": 132846} +{"epoch": 1.13671875, "grad_norm": 0.004269362892955542, "learning_rate": 4.877873600900581e-05, "loss": 5.181954838917591e-05, "memory/device_reserved (GiB)": 35.57, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00005, "step": 291, "tokens/total": 8800368, "tokens/train_per_sec_per_gpu": 35.83, "tokens/trainable": 133347} +{"epoch": 1.140625, "grad_norm": 0.0013308345805853605, "learning_rate": 4.849136411748306e-05, "loss": 2.814647086779587e-05, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00003, "step": 292, "tokens/total": 8830688, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 133779} +{"epoch": 1.14453125, "grad_norm": 0.002314274897798896, "learning_rate": 4.8204263076866574e-05, "loss": 3.513288538670167e-05, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00004, "step": 293, "tokens/total": 8861008, "tokens/train_per_sec_per_gpu": 36.35, "tokens/trainable": 134251} +{"epoch": 1.1484375, "grad_norm": 0.0008702389895915985, "learning_rate": 4.791744483460251e-05, "loss": 2.252616104669869e-05, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00002, "step": 294, "tokens/total": 8891760, "tokens/train_per_sec_per_gpu": 33.62, "tokens/trainable": 134683} +{"epoch": 1.15234375, "grad_norm": 0.018964840099215508, "learning_rate": 4.7630921326368736e-05, "loss": 0.0001443990768166259, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00014, "step": 295, "tokens/total": 8922192, "tokens/train_per_sec_per_gpu": 35.66, "tokens/trainable": 135165} +{"epoch": 1.15625, "grad_norm": 0.011429708451032639, "learning_rate": 4.7344704475577916e-05, "loss": 8.412318129558116e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00008, "step": 296, "tokens/total": 8952496, "tokens/train_per_sec_per_gpu": 33.37, "tokens/trainable": 135599} +{"epoch": 1.16015625, "grad_norm": 0.12274184077978134, "learning_rate": 4.705880619288153e-05, "loss": 0.0006645542453043163, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00066, "step": 297, "tokens/total": 8982768, "tokens/train_per_sec_per_gpu": 32.75, "tokens/trainable": 136038} +{"epoch": 1.1640625, "grad_norm": 0.003483664011582732, "learning_rate": 4.677323837567412e-05, "loss": 6.217554619070143e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00006, "step": 298, "tokens/total": 9013248, "tokens/train_per_sec_per_gpu": 32.99, "tokens/trainable": 136495} +{"epoch": 1.16796875, "grad_norm": 0.0016517788171768188, "learning_rate": 4.6488012907598146e-05, "loss": 3.058795482502319e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00003, "step": 299, "tokens/total": 9043632, "tokens/train_per_sec_per_gpu": 35.43, "tokens/trainable": 136953} +{"epoch": 1.171875, "grad_norm": 0.3304859697818756, "learning_rate": 4.620314165804964e-05, "loss": 0.002893730765208602, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.0029, "step": 300, "tokens/total": 9073936, "tokens/train_per_sec_per_gpu": 34.3, "tokens/trainable": 137412} +{"epoch": 1.17578125, "grad_norm": 0.13385294377803802, "learning_rate": 4.591863648168407e-05, "loss": 0.0007642972050234675, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00076, "step": 301, "tokens/total": 9104416, "tokens/train_per_sec_per_gpu": 35.66, "tokens/trainable": 137907} +{"epoch": 1.1796875, "grad_norm": 0.0037263180129230022, "learning_rate": 4.5634509217923135e-05, "loss": 5.128432167111896e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00005, "step": 302, "tokens/total": 9134688, "tokens/train_per_sec_per_gpu": 34.39, "tokens/trainable": 138383} +{"epoch": 1.18359375, "grad_norm": 0.006775988731533289, "learning_rate": 4.535077169046201e-05, "loss": 2.8198699510539882e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00003, "step": 303, "tokens/total": 9165088, "tokens/train_per_sec_per_gpu": 36.55, "tokens/trainable": 138842} +{"epoch": 1.1875, "grad_norm": 0.004235657397657633, "learning_rate": 4.506743570677743e-05, "loss": 4.3131229176651686e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00004, "step": 304, "tokens/total": 9195584, "tokens/train_per_sec_per_gpu": 30.74, "tokens/trainable": 139269} +{"epoch": 1.19140625, "grad_norm": 0.0024381300900131464, "learning_rate": 4.478451305763618e-05, "loss": 4.280401481082663e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00004, "step": 305, "tokens/total": 9225760, "tokens/train_per_sec_per_gpu": 34.45, "tokens/trainable": 139736} +{"epoch": 1.1953125, "grad_norm": 0.0007152331527322531, "learning_rate": 4.450201551660454e-05, "loss": 1.5340305253630504e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00002, "step": 306, "tokens/total": 9256352, "tokens/train_per_sec_per_gpu": 37.15, "tokens/trainable": 140226} +{"epoch": 1.19921875, "grad_norm": 0.41952189803123474, "learning_rate": 4.4219954839558276e-05, "loss": 0.005223053507506847, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.19, "memory/max_allocated (GiB)": 33.19, "ppl": 1.00524, "step": 307, "tokens/total": 9284256, "tokens/train_per_sec_per_gpu": 34.1, "tokens/trainable": 140664} +{"epoch": 1.203125, "grad_norm": 0.0014089028118178248, "learning_rate": 4.393834276419352e-05, "loss": 2.587199560366571e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00003, "step": 308, "tokens/total": 9314768, "tokens/train_per_sec_per_gpu": 30.02, "tokens/trainable": 141115} +{"epoch": 1.20703125, "grad_norm": 0.5370892882347107, "learning_rate": 4.36571910095382e-05, "loss": 0.002367620589211583, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00237, "step": 309, "tokens/total": 9344976, "tokens/train_per_sec_per_gpu": 29.03, "tokens/trainable": 141530} +{"epoch": 1.2109375, "grad_norm": 0.001014610636048019, "learning_rate": 4.337651127546448e-05, "loss": 1.8852246284950525e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00002, "step": 310, "tokens/total": 9375280, "tokens/train_per_sec_per_gpu": 34.29, "tokens/trainable": 141964} +{"epoch": 1.21484375, "grad_norm": 0.011256729252636433, "learning_rate": 4.3096315242201736e-05, "loss": 6.678313366137445e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00007, "step": 311, "tokens/total": 9405536, "tokens/train_per_sec_per_gpu": 34.0, "tokens/trainable": 142428} +{"epoch": 1.21875, "grad_norm": 0.004441538825631142, "learning_rate": 4.2816614569850635e-05, "loss": 5.578820127993822e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00006, "step": 312, "tokens/total": 9435696, "tokens/train_per_sec_per_gpu": 38.29, "tokens/trainable": 142894} +{"epoch": 1.22265625, "grad_norm": 0.013733429834246635, "learning_rate": 4.2537420897897864e-05, "loss": 0.0001298388233408332, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00013, "step": 313, "tokens/total": 9466080, "tokens/train_per_sec_per_gpu": 36.49, "tokens/trainable": 143368} +{"epoch": 1.2265625, "grad_norm": 0.46937739849090576, "learning_rate": 4.225874584473174e-05, "loss": 0.008783280849456787, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00882, "step": 314, "tokens/total": 9496400, "tokens/train_per_sec_per_gpu": 33.0, "tokens/trainable": 143824} +{"epoch": 1.23046875, "grad_norm": 0.0013688476756215096, "learning_rate": 4.19806010071587e-05, "loss": 2.1897705664741807e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00002, "step": 315, "tokens/total": 9526880, "tokens/train_per_sec_per_gpu": 34.69, "tokens/trainable": 144306} +{"epoch": 1.234375, "grad_norm": 0.0007963742245920002, "learning_rate": 4.170299795992081e-05, "loss": 1.3051090718363412e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00001, "step": 316, "tokens/total": 9557280, "tokens/train_per_sec_per_gpu": 35.54, "tokens/trainable": 144763} +{"epoch": 1.23828125, "grad_norm": 0.002978365868330002, "learning_rate": 4.142594825521398e-05, "loss": 2.2917138267075643e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00002, "step": 317, "tokens/total": 9587712, "tokens/train_per_sec_per_gpu": 37.61, "tokens/trainable": 145255} +{"epoch": 1.2421875, "grad_norm": 0.0005564937018789351, "learning_rate": 4.114946342220728e-05, "loss": 1.2049408724124078e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00001, "step": 318, "tokens/total": 9618064, "tokens/train_per_sec_per_gpu": 36.82, "tokens/trainable": 145714} +{"epoch": 1.24609375, "grad_norm": 0.4178679287433624, "learning_rate": 4.087355496656321e-05, "loss": 0.008204679936170578, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00824, "step": 319, "tokens/total": 9648336, "tokens/train_per_sec_per_gpu": 36.15, "tokens/trainable": 146200} +{"epoch": 1.25, "grad_norm": 0.21265174448490143, "learning_rate": 4.05982343699588e-05, "loss": 0.0010491310385987163, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00105, "step": 320, "tokens/total": 9678688, "tokens/train_per_sec_per_gpu": 34.41, "tokens/trainable": 146668} +{"epoch": 1.25390625, "grad_norm": 0.4667704999446869, "learning_rate": 4.0323513089607876e-05, "loss": 0.007591300178319216, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00762, "step": 321, "tokens/total": 9708848, "tokens/train_per_sec_per_gpu": 35.45, "tokens/trainable": 147114} +{"epoch": 1.2578125, "grad_norm": 0.40933945775032043, "learning_rate": 4.004940255778431e-05, "loss": 0.011366574093699455, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.01143, "step": 322, "tokens/total": 9739360, "tokens/train_per_sec_per_gpu": 35.32, "tokens/trainable": 147564} +{"epoch": 1.26171875, "grad_norm": 0.03840261325240135, "learning_rate": 3.977591418134619e-05, "loss": 0.00024993589613586664, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00025, "step": 323, "tokens/total": 9769776, "tokens/train_per_sec_per_gpu": 34.34, "tokens/trainable": 148013} +{"epoch": 1.265625, "grad_norm": 0.00922832265496254, "learning_rate": 3.95030593412612e-05, "loss": 6.446735642384738e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.01, "memory/max_allocated (GiB)": 34.01, "ppl": 1.00006, "step": 324, "tokens/total": 9800096, "tokens/train_per_sec_per_gpu": 38.19, "tokens/trainable": 148470} +{"epoch": 1.26953125, "grad_norm": 0.007167867384850979, "learning_rate": 3.923084939213296e-05, "loss": 0.0001015613743220456, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0001, "step": 325, "tokens/total": 9830304, "tokens/train_per_sec_per_gpu": 36.64, "tokens/trainable": 148926} +{"epoch": 1.2734375, "grad_norm": 0.006592086516320705, "learning_rate": 3.895929566172861e-05, "loss": 4.839149187318981e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00005, "step": 326, "tokens/total": 9860608, "tokens/train_per_sec_per_gpu": 37.9, "tokens/trainable": 149420} +{"epoch": 1.27734375, "grad_norm": 0.2736883759498596, "learning_rate": 3.868840945050728e-05, "loss": 0.006358759012073278, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00638, "step": 327, "tokens/total": 9890560, "tokens/train_per_sec_per_gpu": 39.59, "tokens/trainable": 149873} +{"epoch": 1.28125, "grad_norm": 0.1449279934167862, "learning_rate": 3.841820203114995e-05, "loss": 0.0009410807979293168, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00094, "step": 328, "tokens/total": 9920880, "tokens/train_per_sec_per_gpu": 33.9, "tokens/trainable": 150347} +{"epoch": 1.28515625, "grad_norm": 0.009757250547409058, "learning_rate": 3.814868464809027e-05, "loss": 0.00017852694145403802, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00018, "step": 329, "tokens/total": 9951280, "tokens/train_per_sec_per_gpu": 31.94, "tokens/trainable": 150822} +{"epoch": 1.2890625, "grad_norm": 0.03571556136012077, "learning_rate": 3.787986851704667e-05, "loss": 0.0004097867349628359, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00041, "step": 330, "tokens/total": 9981600, "tokens/train_per_sec_per_gpu": 35.22, "tokens/trainable": 151266} +{"epoch": 1.29296875, "grad_norm": 0.05014437809586525, "learning_rate": 3.7611764824555654e-05, "loss": 0.0006581329507753253, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00066, "step": 331, "tokens/total": 10012016, "tokens/train_per_sec_per_gpu": 37.2, "tokens/trainable": 151734} +{"epoch": 1.296875, "grad_norm": 0.15289872884750366, "learning_rate": 3.734438472750619e-05, "loss": 0.0020632329396903515, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00207, "step": 332, "tokens/total": 10042448, "tokens/train_per_sec_per_gpu": 34.16, "tokens/trainable": 152184} +{"epoch": 1.30078125, "grad_norm": 0.15180747210979462, "learning_rate": 3.707773935267552e-05, "loss": 0.0012369159376248717, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00124, "step": 333, "tokens/total": 10073024, "tokens/train_per_sec_per_gpu": 31.93, "tokens/trainable": 152623} +{"epoch": 1.3046875, "grad_norm": 0.11855534464120865, "learning_rate": 3.68118397962661e-05, "loss": 0.0021295640617609024, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00213, "step": 334, "tokens/total": 10103504, "tokens/train_per_sec_per_gpu": 33.58, "tokens/trainable": 153040} +{"epoch": 1.30859375, "grad_norm": 0.06385980546474457, "learning_rate": 3.654669712344384e-05, "loss": 0.0008570625213906169, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.31, "memory/max_allocated (GiB)": 33.31, "ppl": 1.00086, "step": 335, "tokens/total": 10131632, "tokens/train_per_sec_per_gpu": 31.9, "tokens/trainable": 153463} +{"epoch": 1.3125, "grad_norm": 0.15227577090263367, "learning_rate": 3.628232236787763e-05, "loss": 0.003443875815719366, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00345, "step": 336, "tokens/total": 10161824, "tokens/train_per_sec_per_gpu": 33.43, "tokens/trainable": 153908} +{"epoch": 1.31640625, "grad_norm": 0.25938382744789124, "learning_rate": 3.6018726531280144e-05, "loss": 0.002705808263272047, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00271, "step": 337, "tokens/total": 10192448, "tokens/train_per_sec_per_gpu": 33.84, "tokens/trainable": 154387} +{"epoch": 1.3203125, "grad_norm": 0.08946357667446136, "learning_rate": 3.575592058295017e-05, "loss": 0.0004575806378852576, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00046, "step": 338, "tokens/total": 10222704, "tokens/train_per_sec_per_gpu": 34.07, "tokens/trainable": 154836} +{"epoch": 1.32421875, "grad_norm": 0.008916565217077732, "learning_rate": 3.549391545931585e-05, "loss": 0.00013453952851705253, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00013, "step": 339, "tokens/total": 10253168, "tokens/train_per_sec_per_gpu": 29.91, "tokens/trainable": 155261} +{"epoch": 1.328125, "grad_norm": 0.006945237051695585, "learning_rate": 3.5232722063479914e-05, "loss": 7.873260619817302e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00008, "step": 340, "tokens/total": 10283552, "tokens/train_per_sec_per_gpu": 31.35, "tokens/trainable": 155674} +{"epoch": 1.33203125, "grad_norm": 0.009922013618052006, "learning_rate": 3.49723512647657e-05, "loss": 0.00010111248411703855, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.0001, "step": 341, "tokens/total": 10313872, "tokens/train_per_sec_per_gpu": 35.89, "tokens/trainable": 156164} +{"epoch": 1.3359375, "grad_norm": 0.02600066177546978, "learning_rate": 3.471281389826491e-05, "loss": 0.00031280232360586524, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00031, "step": 342, "tokens/total": 10344256, "tokens/train_per_sec_per_gpu": 33.72, "tokens/trainable": 156611} +{"epoch": 1.33984375, "grad_norm": 0.016229942440986633, "learning_rate": 3.4454120764386764e-05, "loss": 0.00017198207206092775, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00017, "step": 343, "tokens/total": 10374544, "tokens/train_per_sec_per_gpu": 33.83, "tokens/trainable": 157073} +{"epoch": 1.34375, "grad_norm": 0.016324417665600777, "learning_rate": 3.4196282628408526e-05, "loss": 0.00016402543406002223, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00016, "step": 344, "tokens/total": 10405040, "tokens/train_per_sec_per_gpu": 33.19, "tokens/trainable": 157523} +{"epoch": 1.34765625, "grad_norm": 0.32190635800361633, "learning_rate": 3.3939310220027456e-05, "loss": 0.004134346265345812, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00414, "step": 345, "tokens/total": 10435424, "tokens/train_per_sec_per_gpu": 35.28, "tokens/trainable": 157964} +{"epoch": 1.3515625, "grad_norm": 0.00600312277674675, "learning_rate": 3.3683214232914404e-05, "loss": 8.641595195513219e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00009, "step": 346, "tokens/total": 10465760, "tokens/train_per_sec_per_gpu": 30.55, "tokens/trainable": 158390} +{"epoch": 1.35546875, "grad_norm": 0.13886022567749023, "learning_rate": 3.342800532426873e-05, "loss": 0.0013213125057518482, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00132, "step": 347, "tokens/total": 10495904, "tokens/train_per_sec_per_gpu": 35.72, "tokens/trainable": 158848} +{"epoch": 1.359375, "grad_norm": 0.023262491449713707, "learning_rate": 3.317369411437484e-05, "loss": 0.0002245823125122115, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.0, "memory/max_allocated (GiB)": 34.0, "ppl": 1.00022, "step": 348, "tokens/total": 10526624, "tokens/train_per_sec_per_gpu": 28.78, "tokens/trainable": 159270} +{"epoch": 1.36328125, "grad_norm": 0.208764910697937, "learning_rate": 3.292029118616024e-05, "loss": 0.001788674620911479, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00179, "step": 349, "tokens/total": 10556752, "tokens/train_per_sec_per_gpu": 32.62, "tokens/trainable": 159720} +{"epoch": 1.3671875, "grad_norm": 0.001404186594299972, "learning_rate": 3.266780708475511e-05, "loss": 2.7747140848077834e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00003, "step": 350, "tokens/total": 10587264, "tokens/train_per_sec_per_gpu": 35.27, "tokens/trainable": 160187} +{"epoch": 1.37109375, "grad_norm": 0.012655309401452541, "learning_rate": 3.241625231705354e-05, "loss": 0.0001932395389303565, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00019, "step": 351, "tokens/total": 10617504, "tokens/train_per_sec_per_gpu": 32.63, "tokens/trainable": 160645} +{"epoch": 1.375, "grad_norm": 0.030525848269462585, "learning_rate": 3.216563735127618e-05, "loss": 0.000371504167560488, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00037, "step": 352, "tokens/total": 10647760, "tokens/train_per_sec_per_gpu": 34.64, "tokens/trainable": 161102} +{"epoch": 1.37890625, "grad_norm": 0.0018862849101424217, "learning_rate": 3.191597261653475e-05, "loss": 3.8734902773285285e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00004, "step": 353, "tokens/total": 10678080, "tokens/train_per_sec_per_gpu": 29.02, "tokens/trainable": 161555} +{"epoch": 1.3828125, "grad_norm": 0.0015944828046485782, "learning_rate": 3.166726850239794e-05, "loss": 1.4231935892894398e-05, "memory/device_reserved (GiB)": 35.41, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00001, "step": 354, "tokens/total": 10708544, "tokens/train_per_sec_per_gpu": 36.82, "tokens/trainable": 162020} +{"epoch": 1.38671875, "grad_norm": 0.004869155120104551, "learning_rate": 3.141953535845912e-05, "loss": 7.406627264572307e-05, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00007, "step": 355, "tokens/total": 10739024, "tokens/train_per_sec_per_gpu": 32.21, "tokens/trainable": 162448} +{"epoch": 1.390625, "grad_norm": 0.01870773173868656, "learning_rate": 3.11727834939056e-05, "loss": 0.0001340095914201811, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.61, "memory/max_allocated (GiB)": 33.61, "ppl": 1.00013, "step": 356, "tokens/total": 10769024, "tokens/train_per_sec_per_gpu": 28.32, "tokens/trainable": 162863} +{"epoch": 1.39453125, "grad_norm": 0.08653457462787628, "learning_rate": 3.092702317708967e-05, "loss": 0.0008557374821975827, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00086, "step": 357, "tokens/total": 10799328, "tokens/train_per_sec_per_gpu": 35.09, "tokens/trainable": 163318} +{"epoch": 1.3984375, "grad_norm": 0.49505236744880676, "learning_rate": 3.0682264635101276e-05, "loss": 0.007139122113585472, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00716, "step": 358, "tokens/total": 10829488, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 163767} +{"epoch": 1.40234375, "grad_norm": 0.033166300505399704, "learning_rate": 3.0438518053342407e-05, "loss": 0.00024455596576444805, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00024, "step": 359, "tokens/total": 10859936, "tokens/train_per_sec_per_gpu": 32.09, "tokens/trainable": 164208} +{"epoch": 1.40625, "grad_norm": 0.026151562109589577, "learning_rate": 3.0195793575103266e-05, "loss": 0.00018419645493850112, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00018, "step": 360, "tokens/total": 10890400, "tokens/train_per_sec_per_gpu": 34.29, "tokens/trainable": 164655} +{"epoch": 1.41015625, "grad_norm": 0.5249567627906799, "learning_rate": 2.9954101301140146e-05, "loss": 0.03347098454833031, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.03404, "step": 361, "tokens/total": 10920624, "tokens/train_per_sec_per_gpu": 34.22, "tokens/trainable": 165100} +{"epoch": 1.4140625, "grad_norm": 0.008504888974130154, "learning_rate": 2.9713451289255123e-05, "loss": 0.00012848488404415548, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00013, "step": 362, "tokens/total": 10951136, "tokens/train_per_sec_per_gpu": 35.08, "tokens/trainable": 165587} +{"epoch": 1.41796875, "grad_norm": 0.006400417070835829, "learning_rate": 2.9473853553877484e-05, "loss": 7.728980563115329e-05, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00008, "step": 363, "tokens/total": 10981344, "tokens/train_per_sec_per_gpu": 34.27, "tokens/trainable": 166034} +{"epoch": 1.421875, "grad_norm": 0.010285867378115654, "learning_rate": 2.9235318065647e-05, "loss": 0.00018826585437636822, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00019, "step": 364, "tokens/total": 11011552, "tokens/train_per_sec_per_gpu": 34.21, "tokens/trainable": 166485} +{"epoch": 1.42578125, "grad_norm": 0.05493743345141411, "learning_rate": 2.8997854750998964e-05, "loss": 0.0002043755230261013, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0002, "step": 365, "tokens/total": 11041872, "tokens/train_per_sec_per_gpu": 36.42, "tokens/trainable": 166973} +{"epoch": 1.4296875, "grad_norm": 0.00775935361161828, "learning_rate": 2.8761473491751258e-05, "loss": 0.00012459162098821253, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00012, "step": 366, "tokens/total": 11072192, "tokens/train_per_sec_per_gpu": 29.21, "tokens/trainable": 167414} +{"epoch": 1.43359375, "grad_norm": 0.05490761622786522, "learning_rate": 2.8526184124692883e-05, "loss": 0.000744640186894685, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00074, "step": 367, "tokens/total": 11102736, "tokens/train_per_sec_per_gpu": 30.3, "tokens/trainable": 167851} +{"epoch": 1.4375, "grad_norm": 0.006251805927604437, "learning_rate": 2.829199644117484e-05, "loss": 0.00010329978249501437, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.0001, "step": 368, "tokens/total": 11133056, "tokens/train_per_sec_per_gpu": 35.58, "tokens/trainable": 168294} +{"epoch": 1.44140625, "grad_norm": 0.01030012033879757, "learning_rate": 2.8058920186702553e-05, "loss": 0.0001805450883693993, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00018, "step": 369, "tokens/total": 11163568, "tokens/train_per_sec_per_gpu": 32.32, "tokens/trainable": 168769} +{"epoch": 1.4453125, "grad_norm": 0.007539911661297083, "learning_rate": 2.782696506053033e-05, "loss": 0.00016556130140088499, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00017, "step": 370, "tokens/total": 11193936, "tokens/train_per_sec_per_gpu": 36.96, "tokens/trainable": 169270} +{"epoch": 1.44921875, "grad_norm": 0.004288971424102783, "learning_rate": 2.7596140715257824e-05, "loss": 0.0001009388652164489, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0001, "step": 371, "tokens/total": 11224480, "tokens/train_per_sec_per_gpu": 35.17, "tokens/trainable": 169711} +{"epoch": 1.453125, "grad_norm": 0.01721479743719101, "learning_rate": 2.7366456756428184e-05, "loss": 0.0003073053085245192, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00031, "step": 372, "tokens/total": 11254912, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 170185} +{"epoch": 1.45703125, "grad_norm": 0.008515549823641777, "learning_rate": 2.7137922742128486e-05, "loss": 0.00017193684470839798, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00017, "step": 373, "tokens/total": 11285104, "tokens/train_per_sec_per_gpu": 33.24, "tokens/trainable": 170650} +{"epoch": 1.4609375, "grad_norm": 0.009272439405322075, "learning_rate": 2.691054818259188e-05, "loss": 0.00021900574211031199, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00022, "step": 374, "tokens/total": 11315600, "tokens/train_per_sec_per_gpu": 32.07, "tokens/trainable": 171126} +{"epoch": 1.46484375, "grad_norm": 0.034540578722953796, "learning_rate": 2.6684342539801933e-05, "loss": 0.0005546602769754827, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00055, "step": 375, "tokens/total": 11345776, "tokens/train_per_sec_per_gpu": 35.79, "tokens/trainable": 171600} +{"epoch": 1.46875, "grad_norm": 0.013745410367846489, "learning_rate": 2.645931522709877e-05, "loss": 0.0003212787851225585, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00032, "step": 376, "tokens/total": 11376208, "tokens/train_per_sec_per_gpu": 28.27, "tokens/trainable": 172017} +{"epoch": 1.47265625, "grad_norm": 0.005436223931610584, "learning_rate": 2.6235475608787365e-05, "loss": 0.0001245749299414456, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00012, "step": 377, "tokens/total": 11406512, "tokens/train_per_sec_per_gpu": 36.15, "tokens/trainable": 172493} +{"epoch": 1.4765625, "grad_norm": 0.010458163917064667, "learning_rate": 2.6012832999747916e-05, "loss": 0.0002482631243765354, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00025, "step": 378, "tokens/total": 11436544, "tokens/train_per_sec_per_gpu": 34.54, "tokens/trainable": 172951} +{"epoch": 1.48046875, "grad_norm": 0.006481671240180731, "learning_rate": 2.579139666504821e-05, "loss": 0.0001577583752805367, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00016, "step": 379, "tokens/total": 11467008, "tokens/train_per_sec_per_gpu": 35.79, "tokens/trainable": 173435} +{"epoch": 1.484375, "grad_norm": 0.012174229137599468, "learning_rate": 2.557117581955798e-05, "loss": 0.00027598816086538136, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00028, "step": 380, "tokens/total": 11497184, "tokens/train_per_sec_per_gpu": 30.56, "tokens/trainable": 173900} +{"epoch": 1.48828125, "grad_norm": 0.011468137614428997, "learning_rate": 2.5352179627565532e-05, "loss": 0.00023577807587571442, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00024, "step": 381, "tokens/total": 11527600, "tokens/train_per_sec_per_gpu": 34.28, "tokens/trainable": 174371} +{"epoch": 1.4921875, "grad_norm": 0.005027628969401121, "learning_rate": 2.5134417202396277e-05, "loss": 0.00010462553473189473, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0001, "step": 382, "tokens/total": 11557808, "tokens/train_per_sec_per_gpu": 32.86, "tokens/trainable": 174826} +{"epoch": 1.49609375, "grad_norm": 0.005808121990412474, "learning_rate": 2.491789760603361e-05, "loss": 0.00013498810585588217, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00013, "step": 383, "tokens/total": 11588352, "tokens/train_per_sec_per_gpu": 33.01, "tokens/trainable": 175276} +{"epoch": 1.5, "grad_norm": 0.0358550138771534, "learning_rate": 2.4702629848741764e-05, "loss": 0.0006072560790926218, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00061, "step": 384, "tokens/total": 11618640, "tokens/train_per_sec_per_gpu": 33.79, "tokens/trainable": 175720} +{"epoch": 1.50390625, "grad_norm": 0.09168751537799835, "learning_rate": 2.4488622888690785e-05, "loss": 0.0007024870719760656, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.0007, "step": 385, "tokens/total": 11649072, "tokens/train_per_sec_per_gpu": 35.35, "tokens/trainable": 176203} +{"epoch": 1.5078125, "grad_norm": 0.014628843404352665, "learning_rate": 2.427588563158384e-05, "loss": 0.00020378318731673062, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0002, "step": 386, "tokens/total": 11679552, "tokens/train_per_sec_per_gpu": 34.47, "tokens/trainable": 176651} +{"epoch": 1.51171875, "grad_norm": 0.05326993018388748, "learning_rate": 2.406442693028651e-05, "loss": 0.00035949741140939295, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00036, "step": 387, "tokens/total": 11709760, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 177115} +{"epoch": 1.515625, "grad_norm": 0.008257771842181683, "learning_rate": 2.3854255584458547e-05, "loss": 9.612501162337139e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0001, "step": 388, "tokens/total": 11740288, "tokens/train_per_sec_per_gpu": 36.23, "tokens/trainable": 177601} +{"epoch": 1.51953125, "grad_norm": 0.03791302442550659, "learning_rate": 2.3645380340187508e-05, "loss": 0.0004905672976747155, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00049, "step": 389, "tokens/total": 11770592, "tokens/train_per_sec_per_gpu": 33.09, "tokens/trainable": 178050} +{"epoch": 1.5234375, "grad_norm": 0.2622528076171875, "learning_rate": 2.3437809889624914e-05, "loss": 0.003920107148587704, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00393, "step": 390, "tokens/total": 11800752, "tokens/train_per_sec_per_gpu": 36.56, "tokens/trainable": 178531} +{"epoch": 1.52734375, "grad_norm": 0.06083231046795845, "learning_rate": 2.3231552870624487e-05, "loss": 0.0009291216847486794, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00093, "step": 391, "tokens/total": 11831360, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 179000} +{"epoch": 1.53125, "grad_norm": 0.014715000987052917, "learning_rate": 2.3026617866382657e-05, "loss": 0.00020523657440207899, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00021, "step": 392, "tokens/total": 11861552, "tokens/train_per_sec_per_gpu": 33.21, "tokens/trainable": 179457} +{"epoch": 1.53515625, "grad_norm": 0.01740424521267414, "learning_rate": 2.2823013405081507e-05, "loss": 0.00030176169821061194, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.0003, "step": 393, "tokens/total": 11891904, "tokens/train_per_sec_per_gpu": 32.89, "tokens/trainable": 179890} +{"epoch": 1.5390625, "grad_norm": 0.005717985797673464, "learning_rate": 2.2620747959533722e-05, "loss": 0.00011773040023399517, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00012, "step": 394, "tokens/total": 11922208, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 180341} +{"epoch": 1.54296875, "grad_norm": 0.005951862782239914, "learning_rate": 2.2419829946830123e-05, "loss": 0.0001332889514742419, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00013, "step": 395, "tokens/total": 11952672, "tokens/train_per_sec_per_gpu": 33.1, "tokens/trainable": 180762} +{"epoch": 1.546875, "grad_norm": 0.0560171976685524, "learning_rate": 2.2220267727989325e-05, "loss": 0.00048283624346368015, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00048, "step": 396, "tokens/total": 11983088, "tokens/train_per_sec_per_gpu": 35.3, "tokens/trainable": 181226} +{"epoch": 1.55078125, "grad_norm": 0.013027072884142399, "learning_rate": 2.202206960760984e-05, "loss": 0.00022208130394574255, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00022, "step": 397, "tokens/total": 12013488, "tokens/train_per_sec_per_gpu": 33.28, "tokens/trainable": 181714} +{"epoch": 1.5546875, "grad_norm": 0.0371403768658638, "learning_rate": 2.182524383352446e-05, "loss": 0.0005342152435332537, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00053, "step": 398, "tokens/total": 12043968, "tokens/train_per_sec_per_gpu": 35.68, "tokens/trainable": 182142} +{"epoch": 1.55859375, "grad_norm": 0.004872396122664213, "learning_rate": 2.1629798596457056e-05, "loss": 7.259240373969078e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00007, "step": 399, "tokens/total": 12074160, "tokens/train_per_sec_per_gpu": 34.92, "tokens/trainable": 182604} +{"epoch": 1.5625, "grad_norm": 0.010584473609924316, "learning_rate": 2.1435742029681725e-05, "loss": 0.00017825220129452646, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.67, "memory/max_allocated (GiB)": 33.67, "ppl": 1.00018, "step": 400, "tokens/total": 12104352, "tokens/train_per_sec_per_gpu": 25.33, "tokens/trainable": 183011} +{"epoch": 1.56640625, "grad_norm": 0.007694529369473457, "learning_rate": 2.124308220868431e-05, "loss": 7.509582792408764e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00008, "step": 401, "tokens/total": 12134240, "tokens/train_per_sec_per_gpu": 31.86, "tokens/trainable": 183459} +{"epoch": 1.5703125, "grad_norm": 0.02000581845641136, "learning_rate": 2.105182715082638e-05, "loss": 0.0002455090289004147, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00025, "step": 402, "tokens/total": 12164480, "tokens/train_per_sec_per_gpu": 38.74, "tokens/trainable": 183959} +{"epoch": 1.57421875, "grad_norm": 0.0213518887758255, "learning_rate": 2.0861984815011552e-05, "loss": 0.0002898501115851104, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00029, "step": 403, "tokens/total": 12194640, "tokens/train_per_sec_per_gpu": 32.6, "tokens/trainable": 184431} +{"epoch": 1.578125, "grad_norm": 0.004943408537656069, "learning_rate": 2.0673563101354323e-05, "loss": 0.0001215632728417404, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00012, "step": 404, "tokens/total": 12224960, "tokens/train_per_sec_per_gpu": 31.39, "tokens/trainable": 184867} +{"epoch": 1.58203125, "grad_norm": 0.012947587296366692, "learning_rate": 2.0486569850851317e-05, "loss": 0.0001342954346910119, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00013, "step": 405, "tokens/total": 12255008, "tokens/train_per_sec_per_gpu": 29.46, "tokens/trainable": 185300} +{"epoch": 1.5859375, "grad_norm": 0.05408314988017082, "learning_rate": 2.0301012845054956e-05, "loss": 0.00036081415601074696, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00036, "step": 406, "tokens/total": 12285248, "tokens/train_per_sec_per_gpu": 31.65, "tokens/trainable": 185754} +{"epoch": 1.58984375, "grad_norm": 0.037419769912958145, "learning_rate": 2.011689980574966e-05, "loss": 0.00042080413550138474, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00042, "step": 407, "tokens/total": 12315552, "tokens/train_per_sec_per_gpu": 36.93, "tokens/trainable": 186213} +{"epoch": 1.59375, "grad_norm": 0.035673510283231735, "learning_rate": 1.993423839463052e-05, "loss": 0.000293174380203709, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00029, "step": 408, "tokens/total": 12345904, "tokens/train_per_sec_per_gpu": 32.87, "tokens/trainable": 186659} +{"epoch": 1.59765625, "grad_norm": 0.0039419797249138355, "learning_rate": 1.975303621298445e-05, "loss": 8.023489499464631e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00008, "step": 409, "tokens/total": 12376336, "tokens/train_per_sec_per_gpu": 33.13, "tokens/trainable": 187083} +{"epoch": 1.6015625, "grad_norm": 0.05340631678700447, "learning_rate": 1.957330080137385e-05, "loss": 0.00029110765899531543, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00029, "step": 410, "tokens/total": 12406880, "tokens/train_per_sec_per_gpu": 35.1, "tokens/trainable": 187530} +{"epoch": 1.60546875, "grad_norm": 0.016526181250810623, "learning_rate": 1.9395039639322864e-05, "loss": 0.00015069378423504531, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00015, "step": 411, "tokens/total": 12437088, "tokens/train_per_sec_per_gpu": 34.41, "tokens/trainable": 187972} +{"epoch": 1.609375, "grad_norm": 0.008226566947996616, "learning_rate": 1.9218260145006073e-05, "loss": 9.167128155240789e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00009, "step": 412, "tokens/total": 12465440, "tokens/train_per_sec_per_gpu": 40.08, "tokens/trainable": 188417} +{"epoch": 1.61328125, "grad_norm": 0.005769859533756971, "learning_rate": 1.904296967493982e-05, "loss": 0.00011197607091162354, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00011, "step": 413, "tokens/total": 12495968, "tokens/train_per_sec_per_gpu": 31.63, "tokens/trainable": 188868} +{"epoch": 1.6171875, "grad_norm": 0.08041926473379135, "learning_rate": 1.8869175523676064e-05, "loss": 0.000366955588106066, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00037, "step": 414, "tokens/total": 12526128, "tokens/train_per_sec_per_gpu": 37.02, "tokens/trainable": 189337} +{"epoch": 1.62109375, "grad_norm": 0.03303743153810501, "learning_rate": 1.869688492349885e-05, "loss": 0.0001731862430460751, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00017, "step": 415, "tokens/total": 12556256, "tokens/train_per_sec_per_gpu": 32.0, "tokens/trainable": 189750} +{"epoch": 1.625, "grad_norm": 0.008530828170478344, "learning_rate": 1.85261050441233e-05, "loss": 0.00011680425814120099, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00012, "step": 416, "tokens/total": 12586736, "tokens/train_per_sec_per_gpu": 34.12, "tokens/trainable": 190216} +{"epoch": 1.62890625, "grad_norm": 0.0015226984396576881, "learning_rate": 1.8356842992397304e-05, "loss": 3.2423220545751974e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00003, "step": 417, "tokens/total": 12617168, "tokens/train_per_sec_per_gpu": 35.16, "tokens/trainable": 190708} +{"epoch": 1.6328125, "grad_norm": 0.03927593678236008, "learning_rate": 1.8189105812005714e-05, "loss": 0.00023564108414575458, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00024, "step": 418, "tokens/total": 12647680, "tokens/train_per_sec_per_gpu": 31.01, "tokens/trainable": 191126} +{"epoch": 1.63671875, "grad_norm": 0.06375617533922195, "learning_rate": 1.802290048317732e-05, "loss": 0.0004688606131821871, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00047, "step": 419, "tokens/total": 12677952, "tokens/train_per_sec_per_gpu": 30.44, "tokens/trainable": 191540} +{"epoch": 1.640625, "grad_norm": 0.01704459637403488, "learning_rate": 1.785823392239424e-05, "loss": 0.00010123683023266494, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0001, "step": 420, "tokens/total": 12708416, "tokens/train_per_sec_per_gpu": 36.79, "tokens/trainable": 192014} +{"epoch": 1.64453125, "grad_norm": 0.0012392516946420074, "learning_rate": 1.7695112982104225e-05, "loss": 2.009748641285114e-05, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00002, "step": 421, "tokens/total": 12738640, "tokens/train_per_sec_per_gpu": 31.36, "tokens/trainable": 192463} +{"epoch": 1.6484375, "grad_norm": 0.06435113400220871, "learning_rate": 1.7533544450435433e-05, "loss": 0.0006201984360814095, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00062, "step": 422, "tokens/total": 12769232, "tokens/train_per_sec_per_gpu": 32.83, "tokens/trainable": 192918} +{"epoch": 1.65234375, "grad_norm": 0.04247846081852913, "learning_rate": 1.7373535050913946e-05, "loss": 0.0001703215966699645, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00017, "step": 423, "tokens/total": 12799648, "tokens/train_per_sec_per_gpu": 37.45, "tokens/trainable": 193425} +{"epoch": 1.65625, "grad_norm": 0.0037633716128766537, "learning_rate": 1.721509144218405e-05, "loss": 7.026562525425106e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00007, "step": 424, "tokens/total": 12829920, "tokens/train_per_sec_per_gpu": 35.28, "tokens/trainable": 193898} +{"epoch": 1.66015625, "grad_norm": 0.24774393439292908, "learning_rate": 1.705822021773101e-05, "loss": 0.0018032332882285118, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.0018, "step": 425, "tokens/total": 12860112, "tokens/train_per_sec_per_gpu": 29.32, "tokens/trainable": 194315} +{"epoch": 1.6640625, "grad_norm": 0.0021040684077888727, "learning_rate": 1.69029279056068e-05, "loss": 3.659192589111626e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00004, "step": 426, "tokens/total": 12890320, "tokens/train_per_sec_per_gpu": 35.0, "tokens/trainable": 194796} +{"epoch": 1.66796875, "grad_norm": 0.001372064813040197, "learning_rate": 1.6749220968158415e-05, "loss": 2.2382890165317804e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00002, "step": 427, "tokens/total": 12920656, "tokens/train_per_sec_per_gpu": 34.08, "tokens/trainable": 195262} +{"epoch": 1.671875, "grad_norm": 0.004309098701924086, "learning_rate": 1.659710580175893e-05, "loss": 5.155909457243979e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00005, "step": 428, "tokens/total": 12951040, "tokens/train_per_sec_per_gpu": 31.96, "tokens/trainable": 195680} +{"epoch": 1.67578125, "grad_norm": 0.03517591208219528, "learning_rate": 1.644658873654133e-05, "loss": 0.00021132534311618656, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00021, "step": 429, "tokens/total": 12981232, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 196145} +{"epoch": 1.6796875, "grad_norm": 0.005870590452104807, "learning_rate": 1.629767603613508e-05, "loss": 6.348404713207856e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00006, "step": 430, "tokens/total": 13011616, "tokens/train_per_sec_per_gpu": 32.45, "tokens/trainable": 196626} +{"epoch": 1.68359375, "grad_norm": 0.2625119984149933, "learning_rate": 1.615037389740547e-05, "loss": 0.011768318712711334, "memory/device_reserved (GiB)": 36.35, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.01184, "step": 431, "tokens/total": 13042208, "tokens/train_per_sec_per_gpu": 31.22, "tokens/trainable": 197067} +{"epoch": 1.6875, "grad_norm": 0.007063967175781727, "learning_rate": 1.600468845019576e-05, "loss": 4.784313205163926e-05, "memory/device_reserved (GiB)": 36.35, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00005, "step": 432, "tokens/total": 13072800, "tokens/train_per_sec_per_gpu": 38.79, "tokens/trainable": 197565} +{"epoch": 1.69140625, "grad_norm": 0.006292980629950762, "learning_rate": 1.5860625757072092e-05, "loss": 5.077052628621459e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00005, "step": 433, "tokens/total": 13103328, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 198015} +{"epoch": 1.6953125, "grad_norm": 0.0020431573502719402, "learning_rate": 1.571819181307116e-05, "loss": 3.991067933384329e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00004, "step": 434, "tokens/total": 13133472, "tokens/train_per_sec_per_gpu": 34.03, "tokens/trainable": 198442} +{"epoch": 1.69921875, "grad_norm": 0.004255881067365408, "learning_rate": 1.557739254545075e-05, "loss": 5.153669189894572e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00005, "step": 435, "tokens/total": 13164064, "tokens/train_per_sec_per_gpu": 34.47, "tokens/trainable": 198914} +{"epoch": 1.703125, "grad_norm": 0.46879518032073975, "learning_rate": 1.543823381344311e-05, "loss": 0.013075481168925762, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.01316, "step": 436, "tokens/total": 13194464, "tokens/train_per_sec_per_gpu": 33.5, "tokens/trainable": 199355} +{"epoch": 1.70703125, "grad_norm": 0.22727783024311066, "learning_rate": 1.5300721408011114e-05, "loss": 0.0031720949336886406, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00318, "step": 437, "tokens/total": 13224992, "tokens/train_per_sec_per_gpu": 33.07, "tokens/trainable": 199830} +{"epoch": 1.7109375, "grad_norm": 0.003598392242565751, "learning_rate": 1.5164861051607254e-05, "loss": 5.355676694307476e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00005, "step": 438, "tokens/total": 13255296, "tokens/train_per_sec_per_gpu": 35.02, "tokens/trainable": 200338} +{"epoch": 1.71484375, "grad_norm": 0.04593397676944733, "learning_rate": 1.5030658397935521e-05, "loss": 0.00022313687077257782, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00022, "step": 439, "tokens/total": 13285344, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 200785} +{"epoch": 1.71875, "grad_norm": 0.0011655986309051514, "learning_rate": 1.4898119031716104e-05, "loss": 2.53522812272422e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00003, "step": 440, "tokens/total": 13315632, "tokens/train_per_sec_per_gpu": 36.4, "tokens/trainable": 201267} +{"epoch": 1.72265625, "grad_norm": 0.010816111229360104, "learning_rate": 1.476724846845306e-05, "loss": 0.0002002313849516213, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.0002, "step": 441, "tokens/total": 13346048, "tokens/train_per_sec_per_gpu": 34.31, "tokens/trainable": 201734} +{"epoch": 1.7265625, "grad_norm": 0.008938372135162354, "learning_rate": 1.463805215420471e-05, "loss": 0.00011111146159237251, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00011, "step": 442, "tokens/total": 13376304, "tokens/train_per_sec_per_gpu": 30.45, "tokens/trainable": 202174} +{"epoch": 1.73046875, "grad_norm": 0.010652483440935612, "learning_rate": 1.451053546535705e-05, "loss": 0.000100844117696397, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0001, "step": 443, "tokens/total": 13406720, "tokens/train_per_sec_per_gpu": 36.34, "tokens/trainable": 202612} +{"epoch": 1.734375, "grad_norm": 0.005689944606274366, "learning_rate": 1.438470370840001e-05, "loss": 0.00011003226973116398, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.64, "memory/max_allocated (GiB)": 33.64, "ppl": 1.00011, "step": 444, "tokens/total": 13436464, "tokens/train_per_sec_per_gpu": 31.09, "tokens/trainable": 203020} +{"epoch": 1.73828125, "grad_norm": 0.24016952514648438, "learning_rate": 1.4260562119706606e-05, "loss": 0.002352698938921094, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00236, "step": 445, "tokens/total": 13466672, "tokens/train_per_sec_per_gpu": 34.51, "tokens/trainable": 203479} +{"epoch": 1.7421875, "grad_norm": 0.008882527239620686, "learning_rate": 1.413811586531508e-05, "loss": 0.0001598334638401866, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00016, "step": 446, "tokens/total": 13496736, "tokens/train_per_sec_per_gpu": 35.54, "tokens/trainable": 203946} +{"epoch": 1.74609375, "grad_norm": 0.005755060818046331, "learning_rate": 1.4017370040713884e-05, "loss": 7.665179145988077e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00008, "step": 447, "tokens/total": 13526928, "tokens/train_per_sec_per_gpu": 37.64, "tokens/trainable": 204382} +{"epoch": 1.75, "grad_norm": 0.003990499302744865, "learning_rate": 1.3898329670629645e-05, "loss": 8.206715574488044e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00008, "step": 448, "tokens/total": 13557392, "tokens/train_per_sec_per_gpu": 34.47, "tokens/trainable": 204830} +{"epoch": 1.75390625, "grad_norm": 0.0017717559821903706, "learning_rate": 1.3780999708818058e-05, "loss": 3.886378544848412e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00004, "step": 449, "tokens/total": 13587776, "tokens/train_per_sec_per_gpu": 33.77, "tokens/trainable": 205300} +{"epoch": 1.7578125, "grad_norm": 0.004407316446304321, "learning_rate": 1.3665385037857758e-05, "loss": 6.286771531449631e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00006, "step": 450, "tokens/total": 13618112, "tokens/train_per_sec_per_gpu": 30.74, "tokens/trainable": 205726} +{"epoch": 1.76171875, "grad_norm": 0.004234767984598875, "learning_rate": 1.3551490468947126e-05, "loss": 0.00010631309851305559, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00011, "step": 451, "tokens/total": 13648512, "tokens/train_per_sec_per_gpu": 31.77, "tokens/trainable": 206163} +{"epoch": 1.765625, "grad_norm": 0.19428566098213196, "learning_rate": 1.3439320741704075e-05, "loss": 0.008113588206470013, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00815, "step": 452, "tokens/total": 13678704, "tokens/train_per_sec_per_gpu": 35.55, "tokens/trainable": 206619} +{"epoch": 1.76953125, "grad_norm": 0.35631781816482544, "learning_rate": 1.3328880523968808e-05, "loss": 0.004253438673913479, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00426, "step": 453, "tokens/total": 13709232, "tokens/train_per_sec_per_gpu": 37.36, "tokens/trainable": 207101} +{"epoch": 1.7734375, "grad_norm": 0.004516016226261854, "learning_rate": 1.3220174411609587e-05, "loss": 5.911413245485164e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00006, "step": 454, "tokens/total": 13739600, "tokens/train_per_sec_per_gpu": 35.83, "tokens/trainable": 207546} +{"epoch": 1.77734375, "grad_norm": 0.007509376388043165, "learning_rate": 1.3113206928331471e-05, "loss": 0.00010886647214647382, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00011, "step": 455, "tokens/total": 13769936, "tokens/train_per_sec_per_gpu": 34.43, "tokens/trainable": 207989} +{"epoch": 1.78125, "grad_norm": 0.0016118305502459407, "learning_rate": 1.300798252548806e-05, "loss": 3.1790965294931084e-05, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00003, "step": 456, "tokens/total": 13800240, "tokens/train_per_sec_per_gpu": 34.03, "tokens/trainable": 208402} +{"epoch": 1.78515625, "grad_norm": 0.0032239535357803106, "learning_rate": 1.2904505581896265e-05, "loss": 4.2626015783753246e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00004, "step": 457, "tokens/total": 13830512, "tokens/train_per_sec_per_gpu": 36.74, "tokens/trainable": 208904} +{"epoch": 1.7890625, "grad_norm": 0.0048893350176513195, "learning_rate": 1.2802780403654082e-05, "loss": 6.955669960007071e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00007, "step": 458, "tokens/total": 13860832, "tokens/train_per_sec_per_gpu": 33.78, "tokens/trainable": 209382} +{"epoch": 1.79296875, "grad_norm": 0.003503630170598626, "learning_rate": 1.2702811223961408e-05, "loss": 7.862070197006688e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00008, "step": 459, "tokens/total": 13890992, "tokens/train_per_sec_per_gpu": 33.99, "tokens/trainable": 209833} +{"epoch": 1.796875, "grad_norm": 0.34274882078170776, "learning_rate": 1.2604602202943861e-05, "loss": 0.010280012153089046, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.01033, "step": 460, "tokens/total": 13921424, "tokens/train_per_sec_per_gpu": 32.65, "tokens/trainable": 210284} +{"epoch": 1.80078125, "grad_norm": 0.15473094582557678, "learning_rate": 1.2508157427479686e-05, "loss": 0.0024835069198161364, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.66, "memory/max_allocated (GiB)": 33.66, "ppl": 1.00249, "step": 461, "tokens/total": 13951504, "tokens/train_per_sec_per_gpu": 34.19, "tokens/trainable": 210769} +{"epoch": 1.8046875, "grad_norm": 0.0028502352070063353, "learning_rate": 1.2413480911029655e-05, "loss": 5.3788880904903635e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.38, "memory/max_allocated (GiB)": 33.38, "ppl": 1.00005, "step": 462, "tokens/total": 13979728, "tokens/train_per_sec_per_gpu": 30.54, "tokens/trainable": 211193} +{"epoch": 1.80859375, "grad_norm": 0.0012627877295017242, "learning_rate": 1.2320576593470082e-05, "loss": 3.094038038398139e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00003, "step": 463, "tokens/total": 14009936, "tokens/train_per_sec_per_gpu": 35.23, "tokens/trainable": 211655} +{"epoch": 1.8125, "grad_norm": 0.0036251619458198547, "learning_rate": 1.2229448340928828e-05, "loss": 8.078858081717044e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00008, "step": 464, "tokens/total": 14039872, "tokens/train_per_sec_per_gpu": 29.46, "tokens/trainable": 212085} +{"epoch": 1.81640625, "grad_norm": 0.017260659486055374, "learning_rate": 1.2140099945624458e-05, "loss": 0.0002048999012913555, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0002, "step": 465, "tokens/total": 14070096, "tokens/train_per_sec_per_gpu": 32.77, "tokens/trainable": 212570} +{"epoch": 1.8203125, "grad_norm": 0.2109188288450241, "learning_rate": 1.205253512570841e-05, "loss": 0.006965605076402426, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00699, "step": 466, "tokens/total": 14100192, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 213011} +{"epoch": 1.82421875, "grad_norm": 0.052500564604997635, "learning_rate": 1.1966757525110255e-05, "loss": 0.0004830099060200155, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00048, "step": 467, "tokens/total": 14130448, "tokens/train_per_sec_per_gpu": 31.7, "tokens/trainable": 213450} +{"epoch": 1.828125, "grad_norm": 0.005511680152267218, "learning_rate": 1.1882770713386095e-05, "loss": 8.27932235551998e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00008, "step": 468, "tokens/total": 14160480, "tokens/train_per_sec_per_gpu": 35.81, "tokens/trainable": 213898} +{"epoch": 1.83203125, "grad_norm": 0.14864490926265717, "learning_rate": 1.180057818556998e-05, "loss": 0.0016087992116808891, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00161, "step": 469, "tokens/total": 14191152, "tokens/train_per_sec_per_gpu": 36.95, "tokens/trainable": 214366} +{"epoch": 1.8359375, "grad_norm": 0.004650567192584276, "learning_rate": 1.1720183362028494e-05, "loss": 9.897982818074524e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0001, "step": 470, "tokens/total": 14219392, "tokens/train_per_sec_per_gpu": 35.81, "tokens/trainable": 214800} +{"epoch": 1.83984375, "grad_norm": 0.06575815379619598, "learning_rate": 1.1641589588318387e-05, "loss": 0.0007422211347147822, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00074, "step": 471, "tokens/total": 14249920, "tokens/train_per_sec_per_gpu": 35.59, "tokens/trainable": 215292} +{"epoch": 1.84375, "grad_norm": 0.002274462953209877, "learning_rate": 1.1564800135047418e-05, "loss": 5.863649130333215e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00006, "step": 472, "tokens/total": 14280048, "tokens/train_per_sec_per_gpu": 33.99, "tokens/trainable": 215741} +{"epoch": 1.84765625, "grad_norm": 0.002968342276290059, "learning_rate": 1.148981819773816e-05, "loss": 6.629896233789623e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00007, "step": 473, "tokens/total": 14310432, "tokens/train_per_sec_per_gpu": 34.84, "tokens/trainable": 216194} +{"epoch": 1.8515625, "grad_norm": 0.004850646015256643, "learning_rate": 1.1416646896695086e-05, "loss": 9.518570732325315e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.0001, "step": 474, "tokens/total": 14340496, "tokens/train_per_sec_per_gpu": 33.46, "tokens/trainable": 216634} +{"epoch": 1.85546875, "grad_norm": 0.023391559720039368, "learning_rate": 1.1345289276874717e-05, "loss": 0.00047331867972388864, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00047, "step": 475, "tokens/total": 14370720, "tokens/train_per_sec_per_gpu": 30.6, "tokens/trainable": 217040} +{"epoch": 1.859375, "grad_norm": 0.00271675456315279, "learning_rate": 1.1275748307758873e-05, "loss": 5.663794945576228e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00006, "step": 476, "tokens/total": 14401136, "tokens/train_per_sec_per_gpu": 37.86, "tokens/trainable": 217530} +{"epoch": 1.86328125, "grad_norm": 0.018746392801404, "learning_rate": 1.1208026883231147e-05, "loss": 0.00019438326125964522, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00019, "step": 477, "tokens/total": 14431360, "tokens/train_per_sec_per_gpu": 34.65, "tokens/trainable": 217988} +{"epoch": 1.8671875, "grad_norm": 0.12623320519924164, "learning_rate": 1.1142127821456433e-05, "loss": 0.0016139973886311054, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00162, "step": 478, "tokens/total": 14461968, "tokens/train_per_sec_per_gpu": 32.67, "tokens/trainable": 218461} +{"epoch": 1.87109375, "grad_norm": 0.00717399176210165, "learning_rate": 1.1078053864763674e-05, "loss": 0.00011526358139235526, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00012, "step": 479, "tokens/total": 14491984, "tokens/train_per_sec_per_gpu": 32.93, "tokens/trainable": 218909} +{"epoch": 1.875, "grad_norm": 0.05601891130208969, "learning_rate": 1.1015807679531756e-05, "loss": 0.0003784724394790828, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00038, "step": 480, "tokens/total": 14522544, "tokens/train_per_sec_per_gpu": 34.19, "tokens/trainable": 219350} +{"epoch": 1.87890625, "grad_norm": 0.0026196292601525784, "learning_rate": 1.0955391856078528e-05, "loss": 6.231790757738054e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00006, "step": 481, "tokens/total": 14552560, "tokens/train_per_sec_per_gpu": 29.63, "tokens/trainable": 219776} +{"epoch": 1.8828125, "grad_norm": 0.034271057695150375, "learning_rate": 1.0896808908553007e-05, "loss": 0.00036835690843872726, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00037, "step": 482, "tokens/total": 14583056, "tokens/train_per_sec_per_gpu": 35.19, "tokens/trainable": 220284} +{"epoch": 1.88671875, "grad_norm": 0.38923949003219604, "learning_rate": 1.0840061274830763e-05, "loss": 0.002996535040438175, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.003, "step": 483, "tokens/total": 14613152, "tokens/train_per_sec_per_gpu": 33.17, "tokens/trainable": 220714} +{"epoch": 1.890625, "grad_norm": 0.008355424739420414, "learning_rate": 1.0785151316412473e-05, "loss": 0.00015630274720024318, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00016, "step": 484, "tokens/total": 14643328, "tokens/train_per_sec_per_gpu": 35.08, "tokens/trainable": 221156} +{"epoch": 1.89453125, "grad_norm": 0.0020718539599329233, "learning_rate": 1.0732081318325639e-05, "loss": 4.889669071417302e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00005, "step": 485, "tokens/total": 14673488, "tokens/train_per_sec_per_gpu": 31.7, "tokens/trainable": 221605} +{"epoch": 1.8984375, "grad_norm": 0.005701969377696514, "learning_rate": 1.0680853489029501e-05, "loss": 6.45708350930363e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00006, "step": 486, "tokens/total": 14704000, "tokens/train_per_sec_per_gpu": 30.55, "tokens/trainable": 222042} +{"epoch": 1.90234375, "grad_norm": 0.0030111433006823063, "learning_rate": 1.0631469960323152e-05, "loss": 7.045763777568936e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.67, "memory/max_allocated (GiB)": 33.67, "ppl": 1.00007, "step": 487, "tokens/total": 14733984, "tokens/train_per_sec_per_gpu": 32.15, "tokens/trainable": 222474} +{"epoch": 1.90625, "grad_norm": 0.0025332679506391287, "learning_rate": 1.0583932787256783e-05, "loss": 5.044174031354487e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00005, "step": 488, "tokens/total": 14764400, "tokens/train_per_sec_per_gpu": 36.38, "tokens/trainable": 222967} +{"epoch": 1.91015625, "grad_norm": 0.01333449874073267, "learning_rate": 1.0538243948046206e-05, "loss": 7.783513137837872e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00008, "step": 489, "tokens/total": 14794672, "tokens/train_per_sec_per_gpu": 31.9, "tokens/trainable": 223420} +{"epoch": 1.9140625, "grad_norm": 0.019479215145111084, "learning_rate": 1.0494405343990523e-05, "loss": 0.0001881857169792056, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00019, "step": 490, "tokens/total": 14824784, "tokens/train_per_sec_per_gpu": 29.43, "tokens/trainable": 223864} +{"epoch": 1.91796875, "grad_norm": 0.0038831171113997698, "learning_rate": 1.0452418799392985e-05, "loss": 7.516246841987595e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00008, "step": 491, "tokens/total": 14855056, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 224321} +{"epoch": 1.921875, "grad_norm": 0.015537051483988762, "learning_rate": 1.0412286061485102e-05, "loss": 7.350469240918756e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 34.01, "memory/max_allocated (GiB)": 34.01, "ppl": 1.00007, "step": 492, "tokens/total": 14885392, "tokens/train_per_sec_per_gpu": 36.46, "tokens/trainable": 224802} +{"epoch": 1.92578125, "grad_norm": 0.0036717222537845373, "learning_rate": 1.03740088003539e-05, "loss": 8.332736615557224e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00008, "step": 493, "tokens/total": 14915760, "tokens/train_per_sec_per_gpu": 34.93, "tokens/trainable": 225284} +{"epoch": 1.9296875, "grad_norm": 0.09004666656255722, "learning_rate": 1.0337588608872463e-05, "loss": 0.001120982225984335, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00112, "step": 494, "tokens/total": 14946192, "tokens/train_per_sec_per_gpu": 32.88, "tokens/trainable": 225733} +{"epoch": 1.93359375, "grad_norm": 0.0008198385476134717, "learning_rate": 1.0303027002633622e-05, "loss": 2.1550637029577047e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00002, "step": 495, "tokens/total": 14976272, "tokens/train_per_sec_per_gpu": 32.67, "tokens/trainable": 226172} +{"epoch": 1.9375, "grad_norm": 0.004506978671997786, "learning_rate": 1.0270325419886884e-05, "loss": 5.0953691243194044e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00005, "step": 496, "tokens/total": 15006544, "tokens/train_per_sec_per_gpu": 34.89, "tokens/trainable": 226636} +{"epoch": 1.94140625, "grad_norm": 0.09322372078895569, "learning_rate": 1.0239485221478599e-05, "loss": 0.000592258817050606, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00059, "step": 497, "tokens/total": 15037040, "tokens/train_per_sec_per_gpu": 33.09, "tokens/trainable": 227067} +{"epoch": 1.9453125, "grad_norm": 0.5627581477165222, "learning_rate": 1.0210507690795292e-05, "loss": 0.008054064586758614, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00809, "step": 498, "tokens/total": 15067344, "tokens/train_per_sec_per_gpu": 37.78, "tokens/trainable": 227532} +{"epoch": 1.94921875, "grad_norm": 0.0026798928156495094, "learning_rate": 1.0183394033710305e-05, "loss": 5.531937858904712e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00006, "step": 499, "tokens/total": 15097648, "tokens/train_per_sec_per_gpu": 36.75, "tokens/trainable": 228007} +{"epoch": 1.953125, "grad_norm": 0.04757218807935715, "learning_rate": 1.0158145378533583e-05, "loss": 0.00010594214836601168, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00011, "step": 500, "tokens/total": 15128000, "tokens/train_per_sec_per_gpu": 30.98, "tokens/trainable": 228457} +{"epoch": 1.95703125, "grad_norm": 0.017352212220430374, "learning_rate": 1.0134762775964726e-05, "loss": 0.0002479254035279155, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00025, "step": 501, "tokens/total": 15158352, "tokens/train_per_sec_per_gpu": 35.33, "tokens/trainable": 228928} +{"epoch": 1.9609375, "grad_norm": 0.003859007963910699, "learning_rate": 1.0113247199049278e-05, "loss": 5.532489740289748e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 34.03, "memory/max_allocated (GiB)": 34.03, "ppl": 1.00006, "step": 502, "tokens/total": 15189136, "tokens/train_per_sec_per_gpu": 32.29, "tokens/trainable": 229368} +{"epoch": 1.96484375, "grad_norm": 0.007725914474576712, "learning_rate": 1.0093599543138205e-05, "loss": 0.00010436305456096306, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0001, "step": 503, "tokens/total": 15219616, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 229815} +{"epoch": 1.96875, "grad_norm": 0.0022384016774594784, "learning_rate": 1.0075820625850675e-05, "loss": 5.345371755538508e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00005, "step": 504, "tokens/total": 15250032, "tokens/train_per_sec_per_gpu": 38.98, "tokens/trainable": 230319} +{"epoch": 1.97265625, "grad_norm": 0.003804351668804884, "learning_rate": 1.0059911187040013e-05, "loss": 6.216375186340883e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00006, "step": 505, "tokens/total": 15280592, "tokens/train_per_sec_per_gpu": 37.64, "tokens/trainable": 230807} +{"epoch": 1.9765625, "grad_norm": 0.002130430657416582, "learning_rate": 1.0045871888762893e-05, "loss": 3.884470061166212e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00004, "step": 506, "tokens/total": 15310816, "tokens/train_per_sec_per_gpu": 34.34, "tokens/trainable": 231257} +{"epoch": 1.98046875, "grad_norm": 0.006141587160527706, "learning_rate": 1.003370331525184e-05, "loss": 8.762093784753233e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00009, "step": 507, "tokens/total": 15341424, "tokens/train_per_sec_per_gpu": 36.46, "tokens/trainable": 231746} +{"epoch": 1.984375, "grad_norm": 0.0012427824549376965, "learning_rate": 1.002340597289085e-05, "loss": 3.193254815414548e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.33, "memory/max_allocated (GiB)": 33.33, "ppl": 1.00003, "step": 508, "tokens/total": 15369664, "tokens/train_per_sec_per_gpu": 33.12, "tokens/trainable": 232172} +{"epoch": 1.98828125, "grad_norm": 0.0014760087942704558, "learning_rate": 1.0014980290194387e-05, "loss": 3.683042450575158e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.00004, "step": 509, "tokens/total": 15400496, "tokens/train_per_sec_per_gpu": 37.21, "tokens/trainable": 232680} +{"epoch": 1.9921875, "grad_norm": 0.14180026948451996, "learning_rate": 1.0008426617789489e-05, "loss": 0.0011464458657428622, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00115, "step": 510, "tokens/total": 15430864, "tokens/train_per_sec_per_gpu": 36.21, "tokens/trainable": 233169} +{"epoch": 1.99609375, "grad_norm": 0.007576049771159887, "learning_rate": 1.0003745228401215e-05, "loss": 9.165602386929095e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00009, "step": 511, "tokens/total": 15461232, "tokens/train_per_sec_per_gpu": 32.84, "tokens/trainable": 233609} +{"epoch": 2.0, "grad_norm": 0.0015405165031552315, "learning_rate": 1.0000936316841296e-05, "loss": 3.871887020068243e-05, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00004, "step": 512, "tokens/total": 15491760, "tokens/train_per_sec_per_gpu": 36.36, "tokens/trainable": 234078}