diff --git a/.gitattributes b/.gitattributes index 36c34076ab85d5f635a0416c56c14d0e9004ab30..713cb2b42743d4fd8350e171de42eb3a281052c5 100644 --- a/.gitattributes +++ b/.gitattributes @@ -460,3 +460,20 @@ aft_wave_v2/control_matched__charter0p2/training/checkpoints/checkpoint-512/toke aft_wave_v2/control_matched__charter0p2/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_wave_v2/control_matched__charter0p2/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_wave_v2/control_matched__charter0p2/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/ARTIFACT_MANIFEST.local.json b/aft_wave_v2/coin_real_4x__coin0p2/training/ARTIFACT_MANIFEST.local.json new file mode 100644 index 0000000000000000000000000000000000000000..292e78ab67c2cb9f478c82df7a652a5bb00a1edf --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/ARTIFACT_MANIFEST.local.json @@ -0,0 +1,867 @@ +{ + "repo": "arcadia-impact/scimt-dispatch-models", + "remote_prefix": "aft_wave_v2/coin_real_4x__coin0p2/training", + "local_folder": "/workspace/wave/training", + "files": { + "TRAINED.json": { + "size": 962, + "sha256": "5fa578341331c9d290fb8539f3467c2a0e0874c3611d4344c231b867fe17d5d8" + }, + "axolotl.yaml": { + "size": 1209, + "sha256": "ff0dd0239a28c6e029c5a998e7af1d24f1992fe9ad6c681b2519ea28065459b6" + }, + "checkpoint.json": { + "size": 2208, + "sha256": "f4e2ad56e53297d952a050a601cbbae1a94aa6426e6e9ad5e20bb157cd369cdc" + }, + "checkpoints/README.md": { + "size": 2931, + "sha256": "1d12e1b0bad24e925d77aa4ab1e5971a41fa1395648b0da713fdece8003c0905" + }, + "checkpoints/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/adapter_model.safetensors": { + "size": 547777976, + "sha256": "004757923662cc6573db04eb1edf3bcc6f5152f20b0e388191af7ad75b8ba2f5" + }, + "checkpoints/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-128/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-128/adapter_model.safetensors": { + "size": 547777976, + "sha256": "4fc2babf0e54f7a72a2f3faf84eaa36b6c79b527536e07ce1841f5e43580c86b" + }, + "checkpoints/checkpoint-128/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/optimizer.pt": { + "size": 1048106435, + "sha256": "31ebd596a72dab4299216ee54f745a345beefed2232502312721b27dead61c31" + }, + "checkpoints/checkpoint-128/rng_state.pth": { + "size": 14645, + "sha256": "234e8c0c3fd2b805b594ff492259b2db7aead6c3ca1abbea3168b40935e4d0cb" + }, + "checkpoints/checkpoint-128/scheduler.pt": { + "size": 1465, + "sha256": "efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f" + }, + "checkpoints/checkpoint-128/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-128/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-128/tokens_state.json": { + "size": 38, + "sha256": "5d4e623cef80604dfe57d301601c79f30656bf80c690d125a68a0d329f78f80c" + }, + "checkpoints/checkpoint-128/trainer_state.json": { + "size": 56221, + "sha256": "590883bd12a98ddb9fb206bba6fd1e1db6aa9a48263df74d571086b6a8e94886" + }, + "checkpoints/checkpoint-128/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-160/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-160/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-160/adapter_model.safetensors": { + "size": 547777976, + "sha256": "f18e56b80e7ad24c74c6f7090f4f740933e44ef72100c8a8432894e8c258c433" + }, + "checkpoints/checkpoint-160/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-160/optimizer.pt": { + "size": 1048106435, + "sha256": "07b4ab83e765b9a9fef6d5ad5b0bbf7bcffcc6ee72a12963150438f05716abe2" + }, + "checkpoints/checkpoint-160/rng_state.pth": { + "size": 14645, + "sha256": "62012b9e748bbef1beb75187c13d345f2d456800a4ec5207b243fb40c95ada94" + }, + "checkpoints/checkpoint-160/scheduler.pt": { + "size": 1465, + "sha256": "69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89" + }, + "checkpoints/checkpoint-160/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-160/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-160/tokens_state.json": { + "size": 38, + "sha256": "6d2abd602326bcb4dd54f0732279c81f75489fb06cc9c26b3a1290eac1d64a69" + }, + "checkpoints/checkpoint-160/trainer_state.json": { + "size": 70198, + "sha256": "3fc621af00e4f23772e5259b0934379cabaa9495a4682f38e229c3a9a02c5844" + }, + "checkpoints/checkpoint-160/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-192/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-192/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-192/adapter_model.safetensors": { + "size": 547777976, + "sha256": "85ed8480f6ee209a38ee9cd571b15a1b6c5ce56607a32188e14d48492e058adf" + }, + "checkpoints/checkpoint-192/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-192/optimizer.pt": { + "size": 1048106435, + "sha256": "a86ff874a8c3cc41a595be2ed4adada5891f02ffa60010a45ced596f3483975a" + }, + "checkpoints/checkpoint-192/rng_state.pth": { + "size": 14645, + "sha256": "3d69b2462378ed67f5f5555fb6b192965044d7652528913ba5ccc13c8e15172a" + }, + "checkpoints/checkpoint-192/scheduler.pt": { + "size": 1465, + "sha256": "076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd" + }, + "checkpoints/checkpoint-192/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-192/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-192/tokens_state.json": { + "size": 38, + "sha256": "d0bcbe73caf27bad23d336499deca35036208038fc18e90b69ed1271982cac66" + }, + "checkpoints/checkpoint-192/trainer_state.json": { + "size": 84189, + "sha256": "5689aa913851f531f40d7e32eed5f4e24b92c403457a661581f829bbd513c9c6" + }, + "checkpoints/checkpoint-192/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-224/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-224/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-224/adapter_model.safetensors": { + "size": 547777976, + "sha256": "220fd03ef151db83f45a94e4d3e1a971e762411071355fee57f461304bd52e35" + }, + "checkpoints/checkpoint-224/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-224/optimizer.pt": { + "size": 1048106435, + "sha256": "05c3c1f0a5d74da8e16bfeba10f5ebec805dcc1dce9c651ef3032a4576a630e4" + }, + "checkpoints/checkpoint-224/rng_state.pth": { + "size": 14645, + "sha256": "496ba904d564744600c2dc9631c09eb565ed0719573f403cd75b1e83021c01e1" + }, + "checkpoints/checkpoint-224/scheduler.pt": { + "size": 1465, + "sha256": "f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822" + }, + "checkpoints/checkpoint-224/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-224/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-224/tokens_state.json": { + "size": 39, + "sha256": "837aaa06bfc89ca985eb3d887a0924f1a11a28f8aa3e0d45204320cda4d12b5e" + }, + "checkpoints/checkpoint-224/trainer_state.json": { + "size": 98212, + "sha256": "9cc58420475dae5d789df25bdc999682ac93967119b83fa5a1ef0b4b32889142" + }, + "checkpoints/checkpoint-224/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-256/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-256/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-256/adapter_model.safetensors": { + "size": 547777976, + "sha256": "e2ba4d31dd353bd2ca9294a219e16708e9d8a591bf19e71b06e213df8ed69bea" + }, + "checkpoints/checkpoint-256/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-256/optimizer.pt": { + "size": 1048106435, + "sha256": "eee1b94dd3bdec4d438ef940db0e138077560c7d35ee39467be8687bf23e98bc" + }, + "checkpoints/checkpoint-256/rng_state.pth": { + "size": 14645, + "sha256": "370b4fba725358746db83895d8635e99b2f70cb36421eb17a0433851f5afd453" + }, + "checkpoints/checkpoint-256/scheduler.pt": { + "size": 1465, + "sha256": "1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3" + }, + "checkpoints/checkpoint-256/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-256/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-256/tokens_state.json": { + "size": 39, + "sha256": "d4929d8b8f64e42e8d625a7e18a46658293bb00c21cb89ef543c8423240df68a" + }, + "checkpoints/checkpoint-256/trainer_state.json": { + "size": 112266, + "sha256": "07a486030d1c51e736c694a15ab608049e0a6cd9ad2658e5b040a85a16af4e25" + }, + "checkpoints/checkpoint-256/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-288/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-288/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-288/adapter_model.safetensors": { + "size": 547777976, + "sha256": "d181ae74f1617158aaf65f734801ceae0768304d6027b9e77606e15c50b81dac" + }, + "checkpoints/checkpoint-288/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-288/optimizer.pt": { + "size": 1048106435, + "sha256": "5ca99fac2ee7672f86c7d237de8447789263ea6631bce4dd83c9efe25c9082bd" + }, + "checkpoints/checkpoint-288/rng_state.pth": { + "size": 14645, + "sha256": "47ff3a93a2ca5e4e3e5cc87da30e0868111aa8accfb80c30aa2ec9467da212fb" + }, + "checkpoints/checkpoint-288/scheduler.pt": { + "size": 1465, + "sha256": "a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363" + }, + "checkpoints/checkpoint-288/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-288/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-288/tokens_state.json": { + "size": 39, + "sha256": "af477bcb3b897a3f8f3e8da2dd3a8ebe6cce451be8ae5d737b9cb6ea326ef2d9" + }, + "checkpoints/checkpoint-288/trainer_state.json": { + "size": 126338, + "sha256": "c4c12fb931ec944285efbccc02392b4a02ae98e484bbcc870cd033aac1db78df" + }, + "checkpoints/checkpoint-288/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-32/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-32/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-32/adapter_model.safetensors": { + "size": 547777976, + "sha256": "8e42b5dd58a02736989a8d1a457227b96188ee01f7f469d682169a85ba93d83a" + }, + "checkpoints/checkpoint-32/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-32/optimizer.pt": { + "size": 1048106435, + "sha256": "c29ad0f29115d6a62b344b2719754dd79fc85284138b433bd8db4d102d2bbd2d" + }, + "checkpoints/checkpoint-32/rng_state.pth": { + "size": 14645, + "sha256": "046455cb6ded9b2989f35bf5efbbff0517d6651771b52c93d31759906cafe479" + }, + "checkpoints/checkpoint-32/scheduler.pt": { + "size": 1465, + "sha256": "8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc" + }, + "checkpoints/checkpoint-32/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-32/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-32/tokens_state.json": { + "size": 37, + "sha256": "cec49603be80ef1df2a4dadb25de3e8be8585000fa9c1e62970e67efa8f9d602" + }, + "checkpoints/checkpoint-32/trainer_state.json": { + "size": 14416, + "sha256": "11d44104bbff25df8f8dd6f9724369eed58a187190f359cd22ad52d92b39cd96" + }, + "checkpoints/checkpoint-32/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-320/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-320/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-320/adapter_model.safetensors": { + "size": 547777976, + "sha256": "ce063233756c13a0b13b3e3de54a4f0d7aca4320fdbee2aebc8257408e81990b" + }, + "checkpoints/checkpoint-320/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-320/optimizer.pt": { + "size": 1048106435, + "sha256": "736cb10079d37cfe6e39fc324a83a03cce6bd3dbc2bbadde1513e08d4816b68d" + }, + "checkpoints/checkpoint-320/rng_state.pth": { + "size": 14645, + "sha256": "0391605f7d7b8626c1c97d35086145a3e9e4df5cafc3842162a7664ce0f73f43" + }, + "checkpoints/checkpoint-320/scheduler.pt": { + "size": 1465, + "sha256": "e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0" + }, + "checkpoints/checkpoint-320/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-320/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-320/tokens_state.json": { + "size": 39, + "sha256": "5dba8649e21e024152b4b3daeba662208bb0f33796ac11478c9b0d7cffa65e97" + }, + "checkpoints/checkpoint-320/trainer_state.json": { + "size": 140427, + "sha256": "7ca9473d3142d20df35f85e42aea50f11e14e9d3fc83e5da9ea8351041294c9a" + }, + "checkpoints/checkpoint-320/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-352/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-352/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-352/adapter_model.safetensors": { + "size": 547777976, + "sha256": "43e130a15d3a546161129a78748f0fd0757c0075942046a32d1aa2551e313a6c" + }, + "checkpoints/checkpoint-352/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-352/optimizer.pt": { + "size": 1048106435, + "sha256": "d069a87fce3739148e9df98672687b4e7b8271e1564bda4fa7f13e5ed01c6d04" + }, + "checkpoints/checkpoint-352/rng_state.pth": { + "size": 14645, + "sha256": "a5d3595bca569f2399926c6c523fd98aead95fc6d00353c61841e5028ebbdeff" + }, + "checkpoints/checkpoint-352/scheduler.pt": { + "size": 1465, + "sha256": "153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7" + }, + "checkpoints/checkpoint-352/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-352/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-352/tokens_state.json": { + "size": 40, + "sha256": "8b8f1db9f0311bacf95a403fdc167f8bce05fa9b5631e294453130a08f29b9c4" + }, + "checkpoints/checkpoint-352/trainer_state.json": { + "size": 154545, + "sha256": "5cb5f2bbdb85f774b4c86b35d1ae66b626c1da8ec6d9a6cbda01952fb1096c19" + }, + "checkpoints/checkpoint-352/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-384/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-384/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-384/adapter_model.safetensors": { + "size": 547777976, + "sha256": "19d460cb5f3eef30a27eeda49a98266c4282f6cbae261428e40ebba0ff9681da" + }, + "checkpoints/checkpoint-384/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-384/optimizer.pt": { + "size": 1048106435, + "sha256": "76c47b870ab9f0fa5d81e16be90897cfb5f3c32449a223f59a3d50150c3df02c" + }, + "checkpoints/checkpoint-384/rng_state.pth": { + "size": 14645, + "sha256": "ea55d7f6de0215cad74ef9143c41f13fa0fffc3e1c9c7025b8d1c96ebbb6f69e" + }, + "checkpoints/checkpoint-384/scheduler.pt": { + "size": 1465, + "sha256": "8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79" + }, + "checkpoints/checkpoint-384/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-384/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-384/tokens_state.json": { + "size": 40, + "sha256": "de563327ae6aae9986227588641dbd4a73c35ac9737ce28dd8b18d6807eb841f" + }, + "checkpoints/checkpoint-384/trainer_state.json": { + "size": 168694, + "sha256": "c5c27504df47c098f0c272f420bcd918ca0f1f41d70bfaa1c4ad66e54d2a902d" + }, + "checkpoints/checkpoint-384/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-416/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-416/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-416/adapter_model.safetensors": { + "size": 547777976, + "sha256": "2d0866c11a9e57490eea883b729339cbc59d5006d03a1a4b4a1e1f4a005bd6f5" + }, + "checkpoints/checkpoint-416/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-416/optimizer.pt": { + "size": 1048106435, + "sha256": "9670a3fe3860e7b056f50a7f7c4276e0268cd918775bc811b8a7c463351da4fa" + }, + "checkpoints/checkpoint-416/rng_state.pth": { + "size": 14645, + "sha256": "49f752bd9a9a6e4604278f4cbc2ad9ae381068763e6148938859abe0b809968f" + }, + "checkpoints/checkpoint-416/scheduler.pt": { + "size": 1465, + "sha256": "2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc" + }, + "checkpoints/checkpoint-416/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-416/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-416/tokens_state.json": { + "size": 40, + "sha256": "652b2133eb54c05ce95ac81317114e9b6a0e614398fea583a61e940a40659c7a" + }, + "checkpoints/checkpoint-416/trainer_state.json": { + "size": 182860, + "sha256": "18dca420b73205a3384c6fc22eba4f82d942c2dbc9845f77f40e25262a1b47be" + }, + "checkpoints/checkpoint-416/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-448/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-448/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-448/adapter_model.safetensors": { + "size": 547777976, + "sha256": "8629ea27947d652d3d99b6309d975cd085658fb0bce9b9309f847b13054f480b" + }, + "checkpoints/checkpoint-448/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-448/optimizer.pt": { + "size": 1048106435, + "sha256": "3ba090668a36670fb601cf514daf31a4efeaa3a4e7c505735817d056fd2001d3" + }, + "checkpoints/checkpoint-448/rng_state.pth": { + "size": 14645, + "sha256": "f0f5fea350892f85d3bbab116a9d5db62bdcedcd47807fd37fd8d790fcd8718a" + }, + "checkpoints/checkpoint-448/scheduler.pt": { + "size": 1465, + "sha256": "457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503" + }, + "checkpoints/checkpoint-448/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-448/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-448/tokens_state.json": { + "size": 40, + "sha256": "10af6c079666802d71ae9c72bb1cbc9f22808f4d291cf6e573a5b2e67474c845" + }, + "checkpoints/checkpoint-448/trainer_state.json": { + "size": 197010, + "sha256": "256426ffcb17ece56a204fd53eb57a0b3406aec06e55aa5d8ba680269d6376b5" + }, + "checkpoints/checkpoint-448/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-480/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-480/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-480/adapter_model.safetensors": { + "size": 547777976, + "sha256": "a639f0cd54e96ad33bff59a38f6b3618936c5b19c68f246d12a117646577484c" + }, + "checkpoints/checkpoint-480/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-480/optimizer.pt": { + "size": 1048106435, + "sha256": "eeb38f160641fd181dd0131ae0bbfd45e58e9a572ec1090ba85363fc61743fbb" + }, + "checkpoints/checkpoint-480/rng_state.pth": { + "size": 14645, + "sha256": "6cc653826760fd60176ec4de753ff8ff573e43d06826f127f29d6db37ba7f969" + }, + "checkpoints/checkpoint-480/scheduler.pt": { + "size": 1465, + "sha256": "1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01" + }, + "checkpoints/checkpoint-480/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-480/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-480/tokens_state.json": { + "size": 40, + "sha256": "77c796414a49d4ddc04c6e228fe890623cdead947a9c487fc3d8b63880ca27e4" + }, + "checkpoints/checkpoint-480/trainer_state.json": { + "size": 211173, + "sha256": "133261f11c867ab76d2685388c77365d099769d894d357d752938bee9ac5086e" + }, + "checkpoints/checkpoint-480/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-512/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-512/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-512/adapter_model.safetensors": { + "size": 547777976, + "sha256": "004757923662cc6573db04eb1edf3bcc6f5152f20b0e388191af7ad75b8ba2f5" + }, + "checkpoints/checkpoint-512/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-512/optimizer.pt": { + "size": 1048106435, + "sha256": "65916b4beb82a197b99e16dd2db1ea96ed374c4fc8a805d7a27ee6ac4b30a97a" + }, + "checkpoints/checkpoint-512/rng_state.pth": { + "size": 14645, + "sha256": "75e98dcceb68f1b6ee34a96307625206231cf5a0e0e4e16d8a9eafccc33ea3fe" + }, + "checkpoints/checkpoint-512/scheduler.pt": { + "size": 1465, + "sha256": "697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4" + }, + "checkpoints/checkpoint-512/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-512/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-512/tokens_state.json": { + "size": 40, + "sha256": "fc4826883a470f621364e7e63d0e124d9cfbb8e82cc14c353e3789ba20140b7c" + }, + "checkpoints/checkpoint-512/trainer_state.json": { + "size": 225339, + "sha256": "214dd5dfc66104c85f77093435c580bad895ebea268d82f43a995d8a6de48a2f" + }, + "checkpoints/checkpoint-512/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-64/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-64/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-64/adapter_model.safetensors": { + "size": 547777976, + "sha256": "4b584e3cf8f378cc0d06c9d8e81b75d4ebec54e9ef2e3bc83848da1c57396f4f" + }, + "checkpoints/checkpoint-64/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-64/optimizer.pt": { + "size": 1048106435, + "sha256": "7f8513c58f406f91f4e3f9606a17ac066ff4724ad1843f31a0343e812b8d82fa" + }, + "checkpoints/checkpoint-64/rng_state.pth": { + "size": 14645, + "sha256": "21d00dc563d713980dd7a1d4b126d93ab060fe78b7ba0c4c92f92cd25a992569" + }, + "checkpoints/checkpoint-64/scheduler.pt": { + "size": 1465, + "sha256": "15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503" + }, + "checkpoints/checkpoint-64/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-64/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-64/tokens_state.json": { + "size": 38, + "sha256": "abe2b3d625c28fc391b4b32077194735a0210c1485305bfd5103fbcb44a1be6f" + }, + "checkpoints/checkpoint-64/trainer_state.json": { + "size": 28341, + "sha256": "3c5eb2a6cc04639bfec45f1939f7237f7564c2fd9fcd70698bbe07b72c3a1099" + }, + "checkpoints/checkpoint-64/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-96/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-96/adapter_config.json": { + "size": 1098, + "sha256": "342c73d74da62ac2cef52bcfc2ae0971436571e32b7edd7325cfebfd02306622" + }, + "checkpoints/checkpoint-96/adapter_model.safetensors": { + "size": 547777976, + "sha256": "c9a211aba75f9a340dc57de855c8e2b458e1b586134e0dc1ad3988647f0aa4e9" + }, + "checkpoints/checkpoint-96/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-96/optimizer.pt": { + "size": 1048106435, + "sha256": "8c5a9fb4a808c243ca1d455c97cb546577f69e79453fc34bf8fae30def5fa9a8" + }, + "checkpoints/checkpoint-96/rng_state.pth": { + "size": 14645, + "sha256": "58b2d07d478cf0253d5619ad2e9d23f5e40a65de8495503676e2505cd9d2856a" + }, + "checkpoints/checkpoint-96/scheduler.pt": { + "size": 1465, + "sha256": "786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0" + }, + "checkpoints/checkpoint-96/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-96/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-96/tokens_state.json": { + "size": 38, + "sha256": "0b71c52644fe1614b0f38698396e8137456254e1ccd3cdb601c7d33938f83ceb" + }, + "checkpoints/checkpoint-96/trainer_state.json": { + "size": 42287, + "sha256": "245feb515f34fcec3f22e15c1f38d2ee0048e67dee2620e28321f5b100b120c3" + }, + "checkpoints/checkpoint-96/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/config.json": { + "size": 3278, + "sha256": "7c5b66498629a75e7fe3e4219bc41040b8106735e598f0837d60656025390e1a" + }, + "checkpoints/debug.log": { + "size": 269203, + "sha256": "94ab3fcb79b1485d1dcb2cb4fe45f280ed6de99b12ac070e1d98a73c2314bc0e" + }, + "checkpoints/processor_config.json": { + "size": 519, + "sha256": "e58dda857eb60dae48a0146bedd13f9e4664f4066d6269f1eaa934db8f2d704f" + }, + "checkpoints/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints.jsonl": { + "size": 141, + "sha256": "f9ebd414bb02ac8239ec8648ecf483002b2afe26dfc138486e69202fa52337a6" + }, + "ckpt_coin_real_4x__coin0p2.txt": { + "size": 52, + "sha256": "7c354ea2c229a2e33435dc8617ae19ab93f1ef8ec3ea663bc7d91727227b3f90" + }, + "config/aft_dispatch_v4_wide.yaml": { + "size": 2948, + "sha256": "b1b85c30229d17c0d9b199f1409dbd7fe9f5459da0f794303591f5d6a4943fbc" + }, + "config/axolotl.yaml": { + "size": 1209, + "sha256": "ff0dd0239a28c6e029c5a998e7af1d24f1992fe9ad6c681b2519ea28065459b6" + }, + "health/training_started.json": { + "size": 137, + "sha256": "fe44a4371a7b6185814e095796cfa97d265bfd6ab9823dc26cc9b95b1cc9dd06" + }, + "run.json": { + "size": 405, + "sha256": "b08ec6049e6f3fa743e7c82a8caea668b5dcff1ba81557b0c540eb5136737a4a" + }, + "train.log": { + "size": 276595, + "sha256": "341baeec1d4bc5f5452865434c2d00a0098c829143b4e8cc44cf8ffdea2faa60" + }, + "trainer_state.final.json": { + "size": 225339, + "sha256": "214dd5dfc66104c85f77093435c580bad895ebea268d82f43a995d8a6de48a2f" + }, + "training_examples.jsonl": { + "size": 3159608, + "sha256": "dd0cc64e6ffd90ae6e12011517a011c9dbcebed1a51771f305f8702eb9719e9e" + }, + "training_provenance.json": { + "size": 4086, + "sha256": "823bf968f76b9fd7817bff674e27e152b2e94fd2d06f71456a6d8878233b096d" + }, + "training_trace.jsonl": { + "size": 182071, + "sha256": "880b3973f57663c4fac7f55137ef8c6a9a90f3672d4572f397e946a1488c2cf9" + } + } +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/TRAINED.json b/aft_wave_v2/coin_real_4x__coin0p2/training/TRAINED.json new file mode 100644 index 0000000000000000000000000000000000000000..48a45bac5bd9987a54cbfe71d3d1051e04ca4c97 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/TRAINED.json @@ -0,0 +1,55 @@ +{ + "version": "dispatch_wave_v2", + "arm": "coin_real_4x__coin0p2", + "parameterization": "lora", + "parent_repo": "arcadia-impact/scimt-dispatch-models", + "parent_prefix": "sft_4epoch/coin/checkpoint-48", + "dataset_sha256": "bf34ebe8a30ee23f25ff7e532974b5fc79d0231b4eb0393ba92b50b266c493b9", + "training_rows": 8192, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "minutes": 59.31, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + }, + "optimizer_steps": 512, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "eval_steps": [ + 32, + 64, + 128, + 256, + 512 + ], + "optimizer_state_saved": true +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/axolotl.yaml b/aft_wave_v2/coin_real_4x__coin0p2/training/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74d57a4d285aa1326da135773e4a78dfb12f8047 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/axolotl.yaml @@ -0,0 +1,56 @@ +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_coin0p2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoint.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoint.json new file mode 100644 index 0000000000000000000000000000000000000000..5bea94e76ad65ff5d54cfe52af2b7ebae3c1cea5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoint.json @@ -0,0 +1,74 @@ +{ + "experiment": "scimt-train:coin_real_4x__coin0p2", + "spec": null, + "kind": null, + "note": "Spec-free training stage (scimt.train.train_dataset) \u2014 a post-training link in a staged chain, not a spec install.", + "train": { + "data": "/workspace/wave/data/datasets/aft_coin0p2.jsonl", + "dataset_meta": { + "adhoc": true + }, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "load_checkpoint_path": "/workspace/wave/parent", + "grpo": null, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + } + }, + "run_name": "coin_real_4x__coin0p2", + "pointer_file": "/workspace/wave/training/ckpt_coin_real_4x__coin0p2.txt", + "backend": "axolotl", + "sampler": "/workspace/wave/training/checkpoints/checkpoint-512", + "state": "/workspace/wave/training/checkpoints/checkpoint-512", + "model": "gemma3_12b_it", + "meta": { + "experiment": "scimt-train:coin_real_4x__coin0p2", + "spec": null, + "kind": null, + "note": "Spec-free training stage (scimt.train.train_dataset) \u2014 a post-training link in a staged chain, not a spec install.", + "train": { + "data": "/workspace/wave/data/datasets/aft_coin0p2.jsonl", + "dataset_meta": { + "adhoc": true + }, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "load_checkpoint_path": "/workspace/wave/parent", + "grpo": null, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + } + }, + "run_name": "coin_real_4x__coin0p2", + "pointer_file": "/workspace/wave/training/ckpt_coin_real_4x__coin0p2.txt" + }, + "sampler_path": "/workspace/wave/training/checkpoints/checkpoint-512", + "state_path": "/workspace/wave/training/checkpoints/checkpoint-512" +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints.jsonl b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..0a5fa00d7287f6920786ad43da67f36b8568e07a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints.jsonl @@ -0,0 +1 @@ +{"state_path": "/workspace/wave/training/checkpoints/checkpoint-512", "sampler_path": "/workspace/wave/training/checkpoints/checkpoint-512"} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/README.md new file mode 100644 index 0000000000000000000000000000000000000000..d7f051f23d65efb98b4ad0fddefb353be9558f42 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/README.md @@ -0,0 +1,128 @@ +--- +library_name: peft +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +datasets: +- /workspace/wave/data/datasets/aft_coin0p2.jsonl +pipeline_tag: text-generation +base_model: /workspace/wave/parent +model-index: +- name: workspace/wave/training/checkpoints + results: [] +--- + + + +[Built with Axolotl](https://github.com/axolotl-ai-cloud/axolotl) +
See axolotl config + +axolotl version: `0.17.0` +```yaml +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_coin0p2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj + +``` + +

+ +# workspace/wave/training/checkpoints + +This model was trained from scratch on the /workspace/wave/data/datasets/aft_coin0p2.jsonl dataset. + +## Model description + +More information needed + +## Intended uses & limitations + +More information needed + +## Training and evaluation data + +More information needed + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 0.0001 +- train_batch_size: 16 +- eval_batch_size: 16 +- seed: 42 +- gradient_accumulation_steps: 2 +- total_train_batch_size: 32 +- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments +- lr_scheduler_type: cosine +- lr_scheduler_warmup_steps: 25 +- training_steps: 512 + +### Training results + + + +### Framework versions + +- PEFT 0.19.1 +- Transformers 5.9.0 +- Pytorch 2.12.1+cu126 +- Datasets 4.8.5 +- Tokenizers 0.22.2 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..16470d79c556f350f8eb6fed9179426818a41f86 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:004757923662cc6573db04eb1edf3bcc6f5152f20b0e388191af7ad75b8ba2f5 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..eade51c5bb3a007195b33a2a0de5dd28ca070c12 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4fc2babf0e54f7a72a2f3faf84eaa36b6c79b527536e07ce1841f5e43580c86b +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f2ddafa0add84ef36aa540ff1f8468213de4eea4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:31ebd596a72dab4299216ee54f745a345beefed2232502312721b27dead61c31 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bbe3022faa9624353cf3c3b8807f7005f541a7d0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:234e8c0c3fd2b805b594ff492259b2db7aead6c3ca1abbea3168b40935e4d0cb +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bc9eb8d587c7c4d2a9e80caed8c0953081dce4ba --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a0fef23cd8ad251a64b290efdc3141bd055d1758 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/tokens_state.json @@ -0,0 +1 @@ +{"total": 3872688, "trainable": 58451} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4d18754642ce1afb4dc89a2e434a2f41526642e0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/trainer_state.json @@ -0,0 +1,1826 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5, + "eval_steps": 500, + "global_step": 128, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.6286196555726387e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-128/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..9603ab8b12723efba169155b5a06e9cdb001d4c8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f18e56b80e7ad24c74c6f7090f4f740933e44ef72100c8a8432894e8c258c433 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..75f62f405c743f6e7be06c2418d3a26fe5ec7e0a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:07b4ab83e765b9a9fef6d5ad5b0bbf7bcffcc6ee72a12963150438f05716abe2 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..965105815690df79194d9098b4ca12e6ed656e87 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:62012b9e748bbef1beb75187c13d345f2d456800a4ec5207b243fb40c95ada94 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d47c908eab228a9c923b0bfb0a7fb1a6dfa773e6 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..65a6e4da75ff244ae866232e657a1d65b25d7cd5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/tokens_state.json @@ -0,0 +1 @@ +{"total": 4836224, "trainable": 73058} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2f29e892adec2142adf40e2dcf2febd4a80959d4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/trainer_state.json @@ -0,0 +1,2274 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.625, + "eval_steps": 500, + "global_step": 160, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.2826278453498266e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-160/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..fd04a4e99a444bfea608ae31fa4ae4e47821f9c8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:85ed8480f6ee209a38ee9cd571b15a1b6c5ce56607a32188e14d48492e058adf +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..01f917b24f422f2ea1e859d89d946702dca1bd06 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a86ff874a8c3cc41a595be2ed4adada5891f02ffa60010a45ced596f3483975a +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..8d3d7cc33b02d8d3b3b19df687fcda5bfd1f9627 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3d69b2462378ed67f5f5555fb6b192965044d7652528913ba5ccc13c8e15172a +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..68fafd70d4f03fc26bd72aefbdcd22105c983415 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e89615643e634a782ca178110eba44fc653e053e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/tokens_state.json @@ -0,0 +1 @@ +{"total": 5807168, "trainable": 87734} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..95a35c89acf91dd227fb39656bc9bbd38e0a2b2a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/trainer_state.json @@ -0,0 +1,2722 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.75, + "eval_steps": 500, + "global_step": 192, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.94166427763157e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-192/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..417dc53338a056e92f0d7470b330fb1e6540f863 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:220fd03ef151db83f45a94e4d3e1a971e762411071355fee57f461304bd52e35 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9cc9e91e595401eb2d349245302f0455d32a7e8b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:05c3c1f0a5d74da8e16bfeba10f5ebec805dcc1dce9c651ef3032a4576a630e4 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..b19eba0b2de9e9c1de8b557e0805fd2a97591182 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:496ba904d564744600c2dc9631c09eb565ed0719573f403cd75b1e83021c01e1 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..734e9a25165547994d3add43bbfb2dd65b7e559d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ceb6ddcb0c3ffd3afaa7fa5894bc1443838d4b64 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/tokens_state.json @@ -0,0 +1 @@ +{"total": 6771488, "trainable": 102434} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..870ed099d8961c2c42001c2afa0985fc87be4264 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/trainer_state.json @@ -0,0 +1,3170 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.875, + "eval_steps": 500, + "global_step": 224, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.596204614023711e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-224/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a056a203ac5adbcb22ea10920d1902dd6e51f024 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e2ba4d31dd353bd2ca9294a219e16708e9d8a591bf19e71b06e213df8ed69bea +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..72d8ea0cebd51bb43158a9c80a0836c0c5358bee --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eee1b94dd3bdec4d438ef940db0e138077560c7d35ee39467be8687bf23e98bc +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..efa718f894d322a84dcafebb3dc81deb9c85c9c2 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:370b4fba725358746db83895d8635e99b2f70cb36421eb17a0433851f5afd453 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0d402b407a8fd89f8bcb5de6f509c9979553d18a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f8e50055fd13942ec41e138d8bd20fcfe6c898bd --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/tokens_state.json @@ -0,0 +1 @@ +{"total": 7738608, "trainable": 117038} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..08e2bb3b18ede4b9dbfccc2a815e72d8644a52eb --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/trainer_state.json @@ -0,0 +1,3618 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 256, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.2526454740406835e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-256/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..7ea8541ee2759f53ebc28ef003740e8eb51ba908 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d181ae74f1617158aaf65f734801ceae0768304d6027b9e77606e15c50b81dac +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..1428ed5e3d0f270311e065b1acc6adae43e023b4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5ca99fac2ee7672f86c7d237de8447789263ea6631bce4dd83c9efe25c9082bd +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..2f1456749735c6c3a1a7710ca8772ffe1fb73d2c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47ff3a93a2ca5e4e3e5cc87da30e0868111aa8accfb80c30aa2ec9467da212fb +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5f00d58c65c626a11e09272704f772fedb532412 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2aa6e2769577a41234a80ef23b9d82536fee4c6c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/tokens_state.json @@ -0,0 +1 @@ +{"total": 8709504, "trainable": 131922} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..bd0e65426a6a1f61b59e13a90ab0bde6c92c55cf --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/trainer_state.json @@ -0,0 +1,4066 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.125, + "eval_steps": 500, + "global_step": 288, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.91164932591743e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-288/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a6bb74c07059bb6b581c5a9abb9ae26961791ac2 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8e42b5dd58a02736989a8d1a457227b96188ee01f7f469d682169a85ba93d83a +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..dc7afc71796854fda2f018320aea74798198e5af --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c29ad0f29115d6a62b344b2719754dd79fc85284138b433bd8db4d102d2bbd2d +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bcaad7482cc1813859d1e652f169fa68f16981e3 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:046455cb6ded9b2989f35bf5efbbff0517d6651771b52c93d31759906cafe479 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..02f9ed8e0defc2ebe0d43c0d14abc27c32e2b2cf --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f3bdacd3d072e0a631b89c4037b41375ef42fd59 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/tokens_state.json @@ -0,0 +1 @@ +{"total": 969264, "trainable": 14655} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..bf6aa498e45f03c5ff8a485547dff6622fdfceb0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/trainer_state.json @@ -0,0 +1,482 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.125, + "eval_steps": 500, + "global_step": 32, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.578961181068442e+16, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-32/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0e8a05769aab8e4ec9ca0e42cd9925956ab7883e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce063233756c13a0b13b3e3de54a4f0d7aca4320fdbee2aebc8257408e81990b +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..a8f11a256cecb3aff169f54ff0fa05aa65d5076f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:736cb10079d37cfe6e39fc324a83a03cce6bd3dbc2bbadde1513e08d4816b68d +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..fb2d9796f065a223e6c4ea502c863112b2a15139 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0391605f7d7b8626c1c97d35086145a3e9e4df5cafc3842162a7664ce0f73f43 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..cf4669ac28031073c15cff33e6a2bd7daa0e7337 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..421715dcdca9c91ad96c462d40a80afb871e72f5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/tokens_state.json @@ -0,0 +1 @@ +{"total": 9678688, "trainable": 146668} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0b20c0c53b828c62527824f7052f2c6546fdae88 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/trainer_state.json @@ -0,0 +1,4514 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.25, + "eval_steps": 500, + "global_step": 320, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.569491143349279e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-320/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0461218a52d0ade32bb0c4b8de4fa86c9f714825 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:43e130a15d3a546161129a78748f0fd0757c0075942046a32d1aa2551e313a6c +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..afca831912d51c87f1088872d922cf00d72d437a --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d069a87fce3739148e9df98672687b4e7b8271e1564bda4fa7f13e5ed01c6d04 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..0ae29c6e3264de0b6c52d65d7699ee19a45126df --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a5d3595bca569f2399926c6c523fd98aead95fc6d00353c61841e5028ebbdeff +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..1d2b8d34c008174b230d6dbbee4469d934686b9f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..639c485fed1f2598b14da4a28cd7a2ed978ce7cd --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/tokens_state.json @@ -0,0 +1 @@ +{"total": 10647760, "trainable": 161102} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..634f717fe937ca3eecf2d23e849208f00ce24f46 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/trainer_state.json @@ -0,0 +1,4962 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.375, + "eval_steps": 500, + "global_step": 352, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.018739959225058556, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00018197213648818433, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.22725771367549896, + "learning_rate": 4.004940255778431e-05, + "loss": 0.002088680863380432, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00209, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.000548983458429575, + "learning_rate": 3.977591418134619e-05, + "loss": 1.6327385310432874e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00761087192222476, + "learning_rate": 3.95030593412612e-05, + "loss": 5.408083234215155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00005, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.23, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.008723607286810875, + "learning_rate": 3.923084939213296e-05, + "loss": 8.67105700308457e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006354155018925667, + "learning_rate": 3.895929566172861e-05, + "loss": 5.4418724175775424e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.7132415771484375, + "learning_rate": 3.868840945050728e-05, + "loss": 0.008354030549526215, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00839, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.57, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.0944104865193367, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0006982197519391775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0007, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0028970104176551104, + "learning_rate": 3.814868464809027e-05, + "loss": 3.3767173590604216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00003, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.009064619429409504, + "learning_rate": 3.787986851704667e-05, + "loss": 7.040683703962713e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05380578711628914, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00042552349623292685, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00043, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.00875640194863081, + "learning_rate": 3.734438472750619e-05, + "loss": 6.885632319608703e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00007, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.0009642363293096423, + "learning_rate": 3.707773935267552e-05, + "loss": 2.1829437173437327e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00002, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.0004874366568401456, + "learning_rate": 3.68118397962661e-05, + "loss": 1.1361282304278575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.31265145540237427, + "learning_rate": 3.654669712344384e-05, + "loss": 0.009118539281189442, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00916, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0011828432325273752, + "learning_rate": 3.628232236787763e-05, + "loss": 2.4649223632877693e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.008298320695757866, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00010702587314881384, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.0037000810261815786, + "learning_rate": 3.575592058295017e-05, + "loss": 3.555983494152315e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.30193620920181274, + "learning_rate": 3.549391545931585e-05, + "loss": 0.0002394289622316137, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00024, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.013805638998746872, + "learning_rate": 3.5232722063479914e-05, + "loss": 9.105106437345967e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00009, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.24, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.26545384526252747, + "learning_rate": 3.49723512647657e-05, + "loss": 0.011088686995208263, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01115, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.00922760646790266, + "learning_rate": 3.471281389826491e-05, + "loss": 9.105133358389139e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.14466270804405212, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0014950309414416552, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.0015, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.8, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.10482759773731232, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0009619007469154894, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00096, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.20004107058048248, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.002495028544217348, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0025, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 8.192997932434082, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.006594266276806593, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00662, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.005605250597000122, + "learning_rate": 3.342800532426873e-05, + "loss": 6.323108391370624e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00006, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.003419809276238084, + "learning_rate": 3.317369411437484e-05, + "loss": 5.6915632740128785e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00006, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.85, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.18140868842601776, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0037076647859066725, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00371, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.002614055061712861, + "learning_rate": 3.266780708475511e-05, + "loss": 3.247601125622168e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.002343076979741454, + "learning_rate": 3.241625231705354e-05, + "loss": 4.7991154133342206e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.81, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.03407059609889984, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003904080658685416, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00039, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.71, + "tokens/trainable": 161102 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.227256939836134e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-352/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0fd7e592ea4e400767e5a83633c7d68a42d5efe0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:19d460cb5f3eef30a27eeda49a98266c4282f6cbae261428e40ebba0ff9681da +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..f113fc852968a35afd5eb0a0ecfb24b09ce99bc4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:76c47b870ab9f0fa5d81e16be90897cfb5f3c32449a223f59a3d50150c3df02c +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..5aea99e5163fa108032b2b18ebc587f398e76e51 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea55d7f6de0215cad74ef9143c41f13fa0fffc3e1c9c7025b8d1c96ebbb6f69e +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c676299e10f0344839a2b08f1cc1e004410cac0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..de0b107094014b7da8157ab906e76603a905fde8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/tokens_state.json @@ -0,0 +1 @@ +{"total": 11618640, "trainable": 175720} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..042e929f0a2bdbf88f8fa485995b37e5da691581 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/trainer_state.json @@ -0,0 +1,5410 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.5, + "eval_steps": 500, + "global_step": 384, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.018739959225058556, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00018197213648818433, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.22725771367549896, + "learning_rate": 4.004940255778431e-05, + "loss": 0.002088680863380432, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00209, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.000548983458429575, + "learning_rate": 3.977591418134619e-05, + "loss": 1.6327385310432874e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00761087192222476, + "learning_rate": 3.95030593412612e-05, + "loss": 5.408083234215155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00005, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.23, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.008723607286810875, + "learning_rate": 3.923084939213296e-05, + "loss": 8.67105700308457e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006354155018925667, + "learning_rate": 3.895929566172861e-05, + "loss": 5.4418724175775424e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.7132415771484375, + "learning_rate": 3.868840945050728e-05, + "loss": 0.008354030549526215, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00839, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.57, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.0944104865193367, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0006982197519391775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0007, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0028970104176551104, + "learning_rate": 3.814868464809027e-05, + "loss": 3.3767173590604216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00003, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.009064619429409504, + "learning_rate": 3.787986851704667e-05, + "loss": 7.040683703962713e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05380578711628914, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00042552349623292685, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00043, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.00875640194863081, + "learning_rate": 3.734438472750619e-05, + "loss": 6.885632319608703e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00007, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.0009642363293096423, + "learning_rate": 3.707773935267552e-05, + "loss": 2.1829437173437327e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00002, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.0004874366568401456, + "learning_rate": 3.68118397962661e-05, + "loss": 1.1361282304278575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.31265145540237427, + "learning_rate": 3.654669712344384e-05, + "loss": 0.009118539281189442, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00916, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0011828432325273752, + "learning_rate": 3.628232236787763e-05, + "loss": 2.4649223632877693e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.008298320695757866, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00010702587314881384, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.0037000810261815786, + "learning_rate": 3.575592058295017e-05, + "loss": 3.555983494152315e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.30193620920181274, + "learning_rate": 3.549391545931585e-05, + "loss": 0.0002394289622316137, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00024, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.013805638998746872, + "learning_rate": 3.5232722063479914e-05, + "loss": 9.105106437345967e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00009, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.24, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.26545384526252747, + "learning_rate": 3.49723512647657e-05, + "loss": 0.011088686995208263, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01115, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.00922760646790266, + "learning_rate": 3.471281389826491e-05, + "loss": 9.105133358389139e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.14466270804405212, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0014950309414416552, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.0015, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.8, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.10482759773731232, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0009619007469154894, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00096, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.20004107058048248, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.002495028544217348, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0025, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 8.192997932434082, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.006594266276806593, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00662, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.005605250597000122, + "learning_rate": 3.342800532426873e-05, + "loss": 6.323108391370624e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00006, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.003419809276238084, + "learning_rate": 3.317369411437484e-05, + "loss": 5.6915632740128785e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00006, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.85, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.18140868842601776, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0037076647859066725, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00371, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.002614055061712861, + "learning_rate": 3.266780708475511e-05, + "loss": 3.247601125622168e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.002343076979741454, + "learning_rate": 3.241625231705354e-05, + "loss": 4.7991154133342206e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.81, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.03407059609889984, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003904080658685416, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00039, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.71, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.012719057500362396, + "learning_rate": 3.191597261653475e-05, + "loss": 0.00012954325939062983, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.005359884351491928, + "learning_rate": 3.166726850239794e-05, + "loss": 8.441988029517233e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.056369855999946594, + "learning_rate": 3.141953535845912e-05, + "loss": 0.00027662873617373407, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00028, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0013983466196805239, + "learning_rate": 3.11727834939056e-05, + "loss": 3.224632018827833e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00003, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.48, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.06720387935638428, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0002829490404110402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00028, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.01048748753964901, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.00019201112445443869, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00019, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.11898642778396606, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0006578543689101934, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00066, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.09823726862668991, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.001951234764419496, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00195, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.37214013934135437, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.003241142490878701, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00325, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.033950626850128174, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00024406128795817494, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00024, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.11754101514816284, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.0018770417664200068, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00188, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.012853973545134068, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00021306828421074897, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.09332777559757233, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00026301087928004563, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00026, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.009343368001282215, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.0001053380110533908, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00011, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.035727277398109436, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0005394808831624687, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00054, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.08836446702480316, + "learning_rate": 2.829199644117484e-05, + "loss": 0.000713829998858273, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00071, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.006047180388122797, + "learning_rate": 2.8058920186702553e-05, + "loss": 8.545963646611199e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.21181856095790863, + "learning_rate": 2.782696506053033e-05, + "loss": 0.0027023768052458763, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00271, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001994711346924305, + "learning_rate": 2.7596140715257824e-05, + "loss": 3.8951005990384147e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0069135697558522224, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.00011239978630328551, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00011, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.3907967507839203, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0018560648895800114, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00186, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.0009331480250693858, + "learning_rate": 2.691054818259188e-05, + "loss": 2.2479640392703004e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.22, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.14613144099712372, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0031683319248259068, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00317, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.85, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.00874971691519022, + "learning_rate": 2.645931522709877e-05, + "loss": 7.25445497664623e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.02506718970835209, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.00010988111171172932, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.09279019385576248, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005218038568273187, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.002128554042428732, + "learning_rate": 2.579139666504821e-05, + "loss": 3.309818930574693e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00003, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.08014458417892456, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00045113274245522916, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00045, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.65, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.009996578097343445, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00013630215835291892, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00014, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.0028510303236544132, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.7343612600816414e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.0006682629464194179, + "learning_rate": 2.491789760603361e-05, + "loss": 9.333229172625579e-06, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0072454060427844524, + "learning_rate": 2.4702629848741764e-05, + "loss": 8.043196430662647e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00008, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.88, + "tokens/trainable": 175720 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.886249931577882e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-384/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..872d19dc7a699be29a664b253f0eaf567de87a37 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2d0866c11a9e57490eea883b729339cbc59d5006d03a1a4b4a1e1f4a005bd6f5 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..924b6fd1a87438005580c40a4bcaf2b97056aa9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9670a3fe3860e7b056f50a7f7c4276e0268cd918775bc811b8a7c463351da4fa +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..323da0a5194e5ce925b57124831414c61390f5e0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:49f752bd9a9a6e4604278f4cbc2ad9ae381068763e6148938859abe0b809968f +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..20a981d8c8f57b8d2e18429ceb299414229ad6ac --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..141e4b902aa713abf9a208543b41cda2cb5debfc --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/tokens_state.json @@ -0,0 +1 @@ +{"total": 12586736, "trainable": 190216} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1faa37ccd3ffaa7123c807232eb37f66520658d4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/trainer_state.json @@ -0,0 +1,5858 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.625, + "eval_steps": 500, + "global_step": 416, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.018739959225058556, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00018197213648818433, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.22725771367549896, + "learning_rate": 4.004940255778431e-05, + "loss": 0.002088680863380432, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00209, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.000548983458429575, + "learning_rate": 3.977591418134619e-05, + "loss": 1.6327385310432874e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00761087192222476, + "learning_rate": 3.95030593412612e-05, + "loss": 5.408083234215155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00005, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.23, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.008723607286810875, + "learning_rate": 3.923084939213296e-05, + "loss": 8.67105700308457e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006354155018925667, + "learning_rate": 3.895929566172861e-05, + "loss": 5.4418724175775424e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.7132415771484375, + "learning_rate": 3.868840945050728e-05, + "loss": 0.008354030549526215, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00839, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.57, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.0944104865193367, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0006982197519391775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0007, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0028970104176551104, + "learning_rate": 3.814868464809027e-05, + "loss": 3.3767173590604216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00003, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.009064619429409504, + "learning_rate": 3.787986851704667e-05, + "loss": 7.040683703962713e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05380578711628914, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00042552349623292685, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00043, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.00875640194863081, + "learning_rate": 3.734438472750619e-05, + "loss": 6.885632319608703e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00007, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.0009642363293096423, + "learning_rate": 3.707773935267552e-05, + "loss": 2.1829437173437327e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00002, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.0004874366568401456, + "learning_rate": 3.68118397962661e-05, + "loss": 1.1361282304278575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.31265145540237427, + "learning_rate": 3.654669712344384e-05, + "loss": 0.009118539281189442, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00916, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0011828432325273752, + "learning_rate": 3.628232236787763e-05, + "loss": 2.4649223632877693e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.008298320695757866, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00010702587314881384, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.0037000810261815786, + "learning_rate": 3.575592058295017e-05, + "loss": 3.555983494152315e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.30193620920181274, + "learning_rate": 3.549391545931585e-05, + "loss": 0.0002394289622316137, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00024, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.013805638998746872, + "learning_rate": 3.5232722063479914e-05, + "loss": 9.105106437345967e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00009, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.24, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.26545384526252747, + "learning_rate": 3.49723512647657e-05, + "loss": 0.011088686995208263, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01115, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.00922760646790266, + "learning_rate": 3.471281389826491e-05, + "loss": 9.105133358389139e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.14466270804405212, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0014950309414416552, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.0015, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.8, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.10482759773731232, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0009619007469154894, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00096, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.20004107058048248, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.002495028544217348, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0025, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 8.192997932434082, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.006594266276806593, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00662, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.005605250597000122, + "learning_rate": 3.342800532426873e-05, + "loss": 6.323108391370624e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00006, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.003419809276238084, + "learning_rate": 3.317369411437484e-05, + "loss": 5.6915632740128785e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00006, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.85, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.18140868842601776, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0037076647859066725, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00371, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.002614055061712861, + "learning_rate": 3.266780708475511e-05, + "loss": 3.247601125622168e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.002343076979741454, + "learning_rate": 3.241625231705354e-05, + "loss": 4.7991154133342206e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.81, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.03407059609889984, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003904080658685416, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00039, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.71, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.012719057500362396, + "learning_rate": 3.191597261653475e-05, + "loss": 0.00012954325939062983, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.005359884351491928, + "learning_rate": 3.166726850239794e-05, + "loss": 8.441988029517233e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.056369855999946594, + "learning_rate": 3.141953535845912e-05, + "loss": 0.00027662873617373407, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00028, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0013983466196805239, + "learning_rate": 3.11727834939056e-05, + "loss": 3.224632018827833e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00003, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.48, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.06720387935638428, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0002829490404110402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00028, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.01048748753964901, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.00019201112445443869, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00019, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.11898642778396606, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0006578543689101934, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00066, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.09823726862668991, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.001951234764419496, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00195, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.37214013934135437, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.003241142490878701, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00325, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.033950626850128174, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00024406128795817494, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00024, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.11754101514816284, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.0018770417664200068, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00188, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.012853973545134068, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00021306828421074897, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.09332777559757233, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00026301087928004563, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00026, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.009343368001282215, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.0001053380110533908, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00011, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.035727277398109436, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0005394808831624687, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00054, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.08836446702480316, + "learning_rate": 2.829199644117484e-05, + "loss": 0.000713829998858273, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00071, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.006047180388122797, + "learning_rate": 2.8058920186702553e-05, + "loss": 8.545963646611199e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.21181856095790863, + "learning_rate": 2.782696506053033e-05, + "loss": 0.0027023768052458763, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00271, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001994711346924305, + "learning_rate": 2.7596140715257824e-05, + "loss": 3.8951005990384147e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0069135697558522224, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.00011239978630328551, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00011, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.3907967507839203, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0018560648895800114, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00186, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.0009331480250693858, + "learning_rate": 2.691054818259188e-05, + "loss": 2.2479640392703004e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.22, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.14613144099712372, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0031683319248259068, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00317, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.85, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.00874971691519022, + "learning_rate": 2.645931522709877e-05, + "loss": 7.25445497664623e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.02506718970835209, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.00010988111171172932, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.09279019385576248, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005218038568273187, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.002128554042428732, + "learning_rate": 2.579139666504821e-05, + "loss": 3.309818930574693e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00003, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.08014458417892456, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00045113274245522916, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00045, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.65, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.009996578097343445, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00013630215835291892, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00014, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.0028510303236544132, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.7343612600816414e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.0006682629464194179, + "learning_rate": 2.491789760603361e-05, + "loss": 9.333229172625579e-06, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0072454060427844524, + "learning_rate": 2.4702629848741764e-05, + "loss": 8.043196430662647e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00008, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.88, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.011281410232186317, + "learning_rate": 2.4488622888690785e-05, + "loss": 8.32078221719712e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00008, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.1868448704481125, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0016654160572215915, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00167, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.5, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.0011122338473796844, + "learning_rate": 2.406442693028651e-05, + "loss": 2.4154323909897357e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0008903060806915164, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.0764218788826838e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.004497055895626545, + "learning_rate": 2.3645380340187508e-05, + "loss": 3.050938539672643e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.004634706303477287, + "learning_rate": 2.3437809889624914e-05, + "loss": 4.90621714561712e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.59, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.00251275347545743, + "learning_rate": 2.3231552870624487e-05, + "loss": 4.33354580309242e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.004285548347979784, + "learning_rate": 2.3026617866382657e-05, + "loss": 5.757250255555846e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.08757233619689941, + "learning_rate": 2.2823013405081507e-05, + "loss": 2.351473449380137e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.00247983168810606, + "learning_rate": 2.2620747959533722e-05, + "loss": 4.2569590732455254e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00004, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.0006376878009177744, + "learning_rate": 2.2419829946830123e-05, + "loss": 1.580168100190349e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.010959293693304062, + "learning_rate": 2.2220267727989325e-05, + "loss": 9.101699106395245e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00009, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.0010561353992670774, + "learning_rate": 2.202206960760984e-05, + "loss": 1.6030653569032438e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.025496836751699448, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00019739707931876183, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0002, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.00516202999278903, + "learning_rate": 2.1629798596457056e-05, + "loss": 5.949653859715909e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0005809378926642239, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.5215588973660488e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.36, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0006219783099368215, + "learning_rate": 2.124308220868431e-05, + "loss": 1.3315910109668039e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00001, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0050822533667087555, + "learning_rate": 2.105182715082638e-05, + "loss": 2.660456084413454e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.81, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.004483669530600309, + "learning_rate": 2.0861984815011552e-05, + "loss": 4.43336321040988e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00004, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.002029991941526532, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.8533231670735404e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.0010850686812773347, + "learning_rate": 2.0486569850851317e-05, + "loss": 1.5834324585739523e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00002, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.22020931541919708, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00125799176748842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00126, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.003032378386706114, + "learning_rate": 2.011689980574966e-05, + "loss": 1.8148441085941158e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.01703350804746151, + "learning_rate": 1.993423839463052e-05, + "loss": 2.1378953533712775e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0006738354568369687, + "learning_rate": 1.975303621298445e-05, + "loss": 1.730798976495862e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.0016640514368191361, + "learning_rate": 1.957330080137385e-05, + "loss": 1.8207503671874292e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.05258834734559059, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00014502773410640657, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0004455884627532214, + "learning_rate": 1.9218260145006073e-05, + "loss": 1.1770534911192954e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.06, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.13378198444843292, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0007207246962934732, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.017833322286605835, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00016917653556447476, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00017, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.1, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.0006367648602463305, + "learning_rate": 1.869688492349885e-05, + "loss": 1.2300321031943895e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.00034314877120777965, + "learning_rate": 1.85261050441233e-05, + "loss": 1.0386990652477834e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 190216 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.543353259829796e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-416/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b5c4ad22e359bd531e133c8af260f25ba4dcf206 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8629ea27947d652d3d99b6309d975cd085658fb0bce9b9309f847b13054f480b +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..ef179ad1557d2f8b2aa604e7902577ef82511543 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3ba090668a36670fb601cf514daf31a4efeaa3a4e7c505735817d056fd2001d3 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..a848c2b6c5f74a7dfcb7aeaf75adc3de783f7394 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f0f5fea350892f85d3bbab116a9d5db62bdcedcd47807fd37fd8d790fcd8718a +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..2045249a1135f9f2d87c87c3e02e6227697d0cfa --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c6796d0904681cc1c501bb821a45ca8b7860aa52 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/tokens_state.json @@ -0,0 +1 @@ +{"total": 13557392, "trainable": 204830} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..89e1532327f2cc8cf0ad7f00efd40b34318c3e62 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/trainer_state.json @@ -0,0 +1,6306 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.75, + "eval_steps": 500, + "global_step": 448, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.018739959225058556, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00018197213648818433, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.22725771367549896, + "learning_rate": 4.004940255778431e-05, + "loss": 0.002088680863380432, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00209, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.000548983458429575, + "learning_rate": 3.977591418134619e-05, + "loss": 1.6327385310432874e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00761087192222476, + "learning_rate": 3.95030593412612e-05, + "loss": 5.408083234215155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00005, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.23, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.008723607286810875, + "learning_rate": 3.923084939213296e-05, + "loss": 8.67105700308457e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006354155018925667, + "learning_rate": 3.895929566172861e-05, + "loss": 5.4418724175775424e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.7132415771484375, + "learning_rate": 3.868840945050728e-05, + "loss": 0.008354030549526215, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00839, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.57, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.0944104865193367, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0006982197519391775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0007, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0028970104176551104, + "learning_rate": 3.814868464809027e-05, + "loss": 3.3767173590604216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00003, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.009064619429409504, + "learning_rate": 3.787986851704667e-05, + "loss": 7.040683703962713e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05380578711628914, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00042552349623292685, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00043, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.00875640194863081, + "learning_rate": 3.734438472750619e-05, + "loss": 6.885632319608703e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00007, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.0009642363293096423, + "learning_rate": 3.707773935267552e-05, + "loss": 2.1829437173437327e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00002, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.0004874366568401456, + "learning_rate": 3.68118397962661e-05, + "loss": 1.1361282304278575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.31265145540237427, + "learning_rate": 3.654669712344384e-05, + "loss": 0.009118539281189442, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00916, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0011828432325273752, + "learning_rate": 3.628232236787763e-05, + "loss": 2.4649223632877693e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.008298320695757866, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00010702587314881384, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.0037000810261815786, + "learning_rate": 3.575592058295017e-05, + "loss": 3.555983494152315e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.30193620920181274, + "learning_rate": 3.549391545931585e-05, + "loss": 0.0002394289622316137, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00024, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.013805638998746872, + "learning_rate": 3.5232722063479914e-05, + "loss": 9.105106437345967e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00009, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.24, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.26545384526252747, + "learning_rate": 3.49723512647657e-05, + "loss": 0.011088686995208263, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01115, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.00922760646790266, + "learning_rate": 3.471281389826491e-05, + "loss": 9.105133358389139e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.14466270804405212, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0014950309414416552, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.0015, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.8, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.10482759773731232, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0009619007469154894, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00096, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.20004107058048248, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.002495028544217348, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0025, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 8.192997932434082, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.006594266276806593, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00662, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.005605250597000122, + "learning_rate": 3.342800532426873e-05, + "loss": 6.323108391370624e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00006, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.003419809276238084, + "learning_rate": 3.317369411437484e-05, + "loss": 5.6915632740128785e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00006, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.85, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.18140868842601776, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0037076647859066725, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00371, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.002614055061712861, + "learning_rate": 3.266780708475511e-05, + "loss": 3.247601125622168e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.002343076979741454, + "learning_rate": 3.241625231705354e-05, + "loss": 4.7991154133342206e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.81, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.03407059609889984, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003904080658685416, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00039, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.71, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.012719057500362396, + "learning_rate": 3.191597261653475e-05, + "loss": 0.00012954325939062983, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.005359884351491928, + "learning_rate": 3.166726850239794e-05, + "loss": 8.441988029517233e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.056369855999946594, + "learning_rate": 3.141953535845912e-05, + "loss": 0.00027662873617373407, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00028, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0013983466196805239, + "learning_rate": 3.11727834939056e-05, + "loss": 3.224632018827833e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00003, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.48, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.06720387935638428, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0002829490404110402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00028, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.01048748753964901, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.00019201112445443869, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00019, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.11898642778396606, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0006578543689101934, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00066, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.09823726862668991, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.001951234764419496, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00195, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.37214013934135437, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.003241142490878701, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00325, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.033950626850128174, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00024406128795817494, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00024, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.11754101514816284, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.0018770417664200068, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00188, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.012853973545134068, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00021306828421074897, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.09332777559757233, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00026301087928004563, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00026, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.009343368001282215, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.0001053380110533908, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00011, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.035727277398109436, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0005394808831624687, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00054, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.08836446702480316, + "learning_rate": 2.829199644117484e-05, + "loss": 0.000713829998858273, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00071, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.006047180388122797, + "learning_rate": 2.8058920186702553e-05, + "loss": 8.545963646611199e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.21181856095790863, + "learning_rate": 2.782696506053033e-05, + "loss": 0.0027023768052458763, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00271, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001994711346924305, + "learning_rate": 2.7596140715257824e-05, + "loss": 3.8951005990384147e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0069135697558522224, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.00011239978630328551, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00011, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.3907967507839203, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0018560648895800114, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00186, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.0009331480250693858, + "learning_rate": 2.691054818259188e-05, + "loss": 2.2479640392703004e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.22, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.14613144099712372, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0031683319248259068, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00317, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.85, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.00874971691519022, + "learning_rate": 2.645931522709877e-05, + "loss": 7.25445497664623e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.02506718970835209, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.00010988111171172932, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.09279019385576248, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005218038568273187, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.002128554042428732, + "learning_rate": 2.579139666504821e-05, + "loss": 3.309818930574693e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00003, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.08014458417892456, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00045113274245522916, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00045, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.65, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.009996578097343445, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00013630215835291892, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00014, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.0028510303236544132, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.7343612600816414e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.0006682629464194179, + "learning_rate": 2.491789760603361e-05, + "loss": 9.333229172625579e-06, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0072454060427844524, + "learning_rate": 2.4702629848741764e-05, + "loss": 8.043196430662647e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00008, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.88, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.011281410232186317, + "learning_rate": 2.4488622888690785e-05, + "loss": 8.32078221719712e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00008, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.1868448704481125, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0016654160572215915, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00167, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.5, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.0011122338473796844, + "learning_rate": 2.406442693028651e-05, + "loss": 2.4154323909897357e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0008903060806915164, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.0764218788826838e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.004497055895626545, + "learning_rate": 2.3645380340187508e-05, + "loss": 3.050938539672643e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.004634706303477287, + "learning_rate": 2.3437809889624914e-05, + "loss": 4.90621714561712e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.59, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.00251275347545743, + "learning_rate": 2.3231552870624487e-05, + "loss": 4.33354580309242e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.004285548347979784, + "learning_rate": 2.3026617866382657e-05, + "loss": 5.757250255555846e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.08757233619689941, + "learning_rate": 2.2823013405081507e-05, + "loss": 2.351473449380137e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.00247983168810606, + "learning_rate": 2.2620747959533722e-05, + "loss": 4.2569590732455254e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00004, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.0006376878009177744, + "learning_rate": 2.2419829946830123e-05, + "loss": 1.580168100190349e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.010959293693304062, + "learning_rate": 2.2220267727989325e-05, + "loss": 9.101699106395245e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00009, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.0010561353992670774, + "learning_rate": 2.202206960760984e-05, + "loss": 1.6030653569032438e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.025496836751699448, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00019739707931876183, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0002, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.00516202999278903, + "learning_rate": 2.1629798596457056e-05, + "loss": 5.949653859715909e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0005809378926642239, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.5215588973660488e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.36, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0006219783099368215, + "learning_rate": 2.124308220868431e-05, + "loss": 1.3315910109668039e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00001, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0050822533667087555, + "learning_rate": 2.105182715082638e-05, + "loss": 2.660456084413454e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.81, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.004483669530600309, + "learning_rate": 2.0861984815011552e-05, + "loss": 4.43336321040988e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00004, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.002029991941526532, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.8533231670735404e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.0010850686812773347, + "learning_rate": 2.0486569850851317e-05, + "loss": 1.5834324585739523e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00002, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.22020931541919708, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00125799176748842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00126, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.003032378386706114, + "learning_rate": 2.011689980574966e-05, + "loss": 1.8148441085941158e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.01703350804746151, + "learning_rate": 1.993423839463052e-05, + "loss": 2.1378953533712775e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0006738354568369687, + "learning_rate": 1.975303621298445e-05, + "loss": 1.730798976495862e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.0016640514368191361, + "learning_rate": 1.957330080137385e-05, + "loss": 1.8207503671874292e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.05258834734559059, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00014502773410640657, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0004455884627532214, + "learning_rate": 1.9218260145006073e-05, + "loss": 1.1770534911192954e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.06, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.13378198444843292, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0007207246962934732, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.017833322286605835, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00016917653556447476, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00017, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.1, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.0006367648602463305, + "learning_rate": 1.869688492349885e-05, + "loss": 1.2300321031943895e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.00034314877120777965, + "learning_rate": 1.85261050441233e-05, + "loss": 1.0386990652477834e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 190216 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.11333048343658447, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.00011449151497799903, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00011, + "step": 417, + "tokens/total": 12617168, + "tokens/train_per_sec_per_gpu": 35.18, + "tokens/trainable": 190708 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.002743076765909791, + "learning_rate": 1.8189105812005714e-05, + "loss": 1.5843726941966452e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 418, + "tokens/total": 12647680, + "tokens/train_per_sec_per_gpu": 30.93, + "tokens/trainable": 191126 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.0011538882972672582, + "learning_rate": 1.802290048317732e-05, + "loss": 1.4711402400280349e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 419, + "tokens/total": 12677952, + "tokens/train_per_sec_per_gpu": 30.43, + "tokens/trainable": 191540 + }, + { + "epoch": 1.640625, + "grad_norm": 0.004467409569770098, + "learning_rate": 1.785823392239424e-05, + "loss": 4.823958806809969e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 420, + "tokens/total": 12708416, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.0012163098435848951, + "learning_rate": 1.7695112982104225e-05, + "loss": 1.1310569789202418e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00001, + "step": 421, + "tokens/total": 12738640, + "tokens/train_per_sec_per_gpu": 31.42, + "tokens/trainable": 192463 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.05586014315485954, + "learning_rate": 1.7533544450435433e-05, + "loss": 0.0002712146961130202, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00027, + "step": 422, + "tokens/total": 12769232, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 192918 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.06670165061950684, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.0004127066058572382, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00041, + "step": 423, + "tokens/total": 12799648, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 193425 + }, + { + "epoch": 1.65625, + "grad_norm": 0.019891072064638138, + "learning_rate": 1.721509144218405e-05, + "loss": 0.00023232161765918136, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00023, + "step": 424, + "tokens/total": 12829920, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 193898 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.006559237837791443, + "learning_rate": 1.705822021773101e-05, + "loss": 6.114803545642644e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 425, + "tokens/total": 12860112, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 194315 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.005621429067105055, + "learning_rate": 1.69029279056068e-05, + "loss": 6.088989175623283e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 426, + "tokens/total": 12890320, + "tokens/train_per_sec_per_gpu": 35.06, + "tokens/trainable": 194796 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.0004855124861933291, + "learning_rate": 1.6749220968158415e-05, + "loss": 1.0911244316957891e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00001, + "step": 427, + "tokens/total": 12920656, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 195262 + }, + { + "epoch": 1.671875, + "grad_norm": 0.0006959444144740701, + "learning_rate": 1.659710580175893e-05, + "loss": 1.4198772987583652e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 428, + "tokens/total": 12951040, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 195680 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.00048281854833476245, + "learning_rate": 1.644658873654133e-05, + "loss": 1.0396102879894897e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00001, + "step": 429, + "tokens/total": 12981232, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 196145 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.007472805213183165, + "learning_rate": 1.629767603613508e-05, + "loss": 5.188406430534087e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 430, + "tokens/total": 13011616, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 196626 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.008452537469565868, + "learning_rate": 1.615037389740547e-05, + "loss": 2.078613033518195e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00002, + "step": 431, + "tokens/total": 13042208, + "tokens/train_per_sec_per_gpu": 31.22, + "tokens/trainable": 197067 + }, + { + "epoch": 1.6875, + "grad_norm": 0.0006947954534552991, + "learning_rate": 1.600468845019576e-05, + "loss": 1.0801179087138735e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00001, + "step": 432, + "tokens/total": 13072800, + "tokens/train_per_sec_per_gpu": 38.63, + "tokens/trainable": 197565 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.0020175843965262175, + "learning_rate": 1.5860625757072092e-05, + "loss": 1.564873855386395e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 433, + "tokens/total": 13103328, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 198015 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.001340522081591189, + "learning_rate": 1.571819181307116e-05, + "loss": 1.7318390746368095e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 434, + "tokens/total": 13133472, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 198442 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.08985895663499832, + "learning_rate": 1.557739254545075e-05, + "loss": 0.0004795463755726814, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 435, + "tokens/total": 13164064, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 198914 + }, + { + "epoch": 1.703125, + "grad_norm": 0.0005816713673993945, + "learning_rate": 1.543823381344311e-05, + "loss": 1.203996089316206e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00001, + "step": 436, + "tokens/total": 13194464, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 199355 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.2994021773338318, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.0055721537210047245, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00559, + "step": 437, + "tokens/total": 13224992, + "tokens/train_per_sec_per_gpu": 32.98, + "tokens/trainable": 199830 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.0006708212895318866, + "learning_rate": 1.5164861051607254e-05, + "loss": 9.834290722210426e-06, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00001, + "step": 438, + "tokens/total": 13255296, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 200338 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.004102411679923534, + "learning_rate": 1.5030658397935521e-05, + "loss": 2.766325997072272e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00003, + "step": 439, + "tokens/total": 13285344, + "tokens/train_per_sec_per_gpu": 32.91, + "tokens/trainable": 200785 + }, + { + "epoch": 1.71875, + "grad_norm": 0.0009204484522342682, + "learning_rate": 1.4898119031716104e-05, + "loss": 1.4549179468303919e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00001, + "step": 440, + "tokens/total": 13315632, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 201267 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.0014299805043265224, + "learning_rate": 1.476724846845306e-05, + "loss": 2.365603722864762e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 441, + "tokens/total": 13346048, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 201734 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.01532546803355217, + "learning_rate": 1.463805215420471e-05, + "loss": 9.99852636596188e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 442, + "tokens/total": 13376304, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 202174 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.18507054448127747, + "learning_rate": 1.451053546535705e-05, + "loss": 0.000987955485470593, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00099, + "step": 443, + "tokens/total": 13406720, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 202612 + }, + { + "epoch": 1.734375, + "grad_norm": 0.028212063014507294, + "learning_rate": 1.438470370840001e-05, + "loss": 0.00025927621754817665, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00026, + "step": 444, + "tokens/total": 13436464, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 203020 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.0004877297324128449, + "learning_rate": 1.4260562119706606e-05, + "loss": 1.2777243682648987e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 445, + "tokens/total": 13466672, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 203479 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.02534927800297737, + "learning_rate": 1.413811586531508e-05, + "loss": 3.3006788726197556e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 446, + "tokens/total": 13496736, + "tokens/train_per_sec_per_gpu": 35.38, + "tokens/trainable": 203946 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.0006437217816710472, + "learning_rate": 1.4017370040713884e-05, + "loss": 1.1569853086257353e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 447, + "tokens/total": 13526928, + "tokens/train_per_sec_per_gpu": 37.38, + "tokens/trainable": 204382 + }, + { + "epoch": 1.75, + "grad_norm": 0.0014389768475666642, + "learning_rate": 1.3898329670629645e-05, + "loss": 2.1063802705612034e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 448, + "tokens/total": 13557392, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 204830 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.202194209681556e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-448/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..ea4d648e2a34d688d30b816f39dac765ba2e3f74 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a639f0cd54e96ad33bff59a38f6b3618936c5b19c68f246d12a117646577484c +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..799d06a99f601dd458d385d45ed134dad964b735 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eeb38f160641fd181dd0131ae0bbfd45e58e9a572ec1090ba85363fc61743fbb +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..94daead8b73acb85d313bc316b68b7b7c8ba7b20 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6cc653826760fd60176ec4de753ff8ff573e43d06826f127f29d6db37ba7f969 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ce1c4e63a6c6f8be8c137d3add4eda184d59a043 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..d024b241b4429724f854cc38e18f3a48b6b05dea --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/tokens_state.json @@ -0,0 +1 @@ +{"total": 14522528, "trainable": 219349} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..094fd51f139886c93cfa9f42c8ca365eb1a43c36 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/trainer_state.json @@ -0,0 +1,6754 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.875, + "eval_steps": 500, + "global_step": 480, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.018739959225058556, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00018197213648818433, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.22725771367549896, + "learning_rate": 4.004940255778431e-05, + "loss": 0.002088680863380432, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00209, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.000548983458429575, + "learning_rate": 3.977591418134619e-05, + "loss": 1.6327385310432874e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00761087192222476, + "learning_rate": 3.95030593412612e-05, + "loss": 5.408083234215155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00005, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.23, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.008723607286810875, + "learning_rate": 3.923084939213296e-05, + "loss": 8.67105700308457e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006354155018925667, + "learning_rate": 3.895929566172861e-05, + "loss": 5.4418724175775424e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.7132415771484375, + "learning_rate": 3.868840945050728e-05, + "loss": 0.008354030549526215, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00839, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.57, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.0944104865193367, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0006982197519391775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0007, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0028970104176551104, + "learning_rate": 3.814868464809027e-05, + "loss": 3.3767173590604216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00003, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.009064619429409504, + "learning_rate": 3.787986851704667e-05, + "loss": 7.040683703962713e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05380578711628914, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00042552349623292685, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00043, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.00875640194863081, + "learning_rate": 3.734438472750619e-05, + "loss": 6.885632319608703e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00007, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.0009642363293096423, + "learning_rate": 3.707773935267552e-05, + "loss": 2.1829437173437327e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00002, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.0004874366568401456, + "learning_rate": 3.68118397962661e-05, + "loss": 1.1361282304278575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.31265145540237427, + "learning_rate": 3.654669712344384e-05, + "loss": 0.009118539281189442, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00916, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0011828432325273752, + "learning_rate": 3.628232236787763e-05, + "loss": 2.4649223632877693e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.008298320695757866, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00010702587314881384, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.0037000810261815786, + "learning_rate": 3.575592058295017e-05, + "loss": 3.555983494152315e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.30193620920181274, + "learning_rate": 3.549391545931585e-05, + "loss": 0.0002394289622316137, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00024, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.013805638998746872, + "learning_rate": 3.5232722063479914e-05, + "loss": 9.105106437345967e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00009, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.24, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.26545384526252747, + "learning_rate": 3.49723512647657e-05, + "loss": 0.011088686995208263, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01115, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.00922760646790266, + "learning_rate": 3.471281389826491e-05, + "loss": 9.105133358389139e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.14466270804405212, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0014950309414416552, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.0015, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.8, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.10482759773731232, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0009619007469154894, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00096, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.20004107058048248, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.002495028544217348, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0025, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 8.192997932434082, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.006594266276806593, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00662, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.005605250597000122, + "learning_rate": 3.342800532426873e-05, + "loss": 6.323108391370624e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00006, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.003419809276238084, + "learning_rate": 3.317369411437484e-05, + "loss": 5.6915632740128785e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00006, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.85, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.18140868842601776, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0037076647859066725, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00371, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.002614055061712861, + "learning_rate": 3.266780708475511e-05, + "loss": 3.247601125622168e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.002343076979741454, + "learning_rate": 3.241625231705354e-05, + "loss": 4.7991154133342206e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.81, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.03407059609889984, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003904080658685416, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00039, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.71, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.012719057500362396, + "learning_rate": 3.191597261653475e-05, + "loss": 0.00012954325939062983, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.005359884351491928, + "learning_rate": 3.166726850239794e-05, + "loss": 8.441988029517233e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.056369855999946594, + "learning_rate": 3.141953535845912e-05, + "loss": 0.00027662873617373407, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00028, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0013983466196805239, + "learning_rate": 3.11727834939056e-05, + "loss": 3.224632018827833e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00003, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.48, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.06720387935638428, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0002829490404110402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00028, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.01048748753964901, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.00019201112445443869, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00019, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.11898642778396606, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0006578543689101934, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00066, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.09823726862668991, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.001951234764419496, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00195, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.37214013934135437, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.003241142490878701, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00325, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.033950626850128174, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00024406128795817494, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00024, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.11754101514816284, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.0018770417664200068, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00188, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.012853973545134068, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00021306828421074897, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.09332777559757233, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00026301087928004563, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00026, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.009343368001282215, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.0001053380110533908, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00011, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.035727277398109436, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0005394808831624687, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00054, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.08836446702480316, + "learning_rate": 2.829199644117484e-05, + "loss": 0.000713829998858273, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00071, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.006047180388122797, + "learning_rate": 2.8058920186702553e-05, + "loss": 8.545963646611199e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.21181856095790863, + "learning_rate": 2.782696506053033e-05, + "loss": 0.0027023768052458763, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00271, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001994711346924305, + "learning_rate": 2.7596140715257824e-05, + "loss": 3.8951005990384147e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0069135697558522224, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.00011239978630328551, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00011, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.3907967507839203, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0018560648895800114, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00186, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.0009331480250693858, + "learning_rate": 2.691054818259188e-05, + "loss": 2.2479640392703004e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.22, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.14613144099712372, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0031683319248259068, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00317, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.85, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.00874971691519022, + "learning_rate": 2.645931522709877e-05, + "loss": 7.25445497664623e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.02506718970835209, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.00010988111171172932, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.09279019385576248, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005218038568273187, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.002128554042428732, + "learning_rate": 2.579139666504821e-05, + "loss": 3.309818930574693e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00003, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.08014458417892456, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00045113274245522916, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00045, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.65, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.009996578097343445, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00013630215835291892, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00014, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.0028510303236544132, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.7343612600816414e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.0006682629464194179, + "learning_rate": 2.491789760603361e-05, + "loss": 9.333229172625579e-06, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0072454060427844524, + "learning_rate": 2.4702629848741764e-05, + "loss": 8.043196430662647e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00008, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.88, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.011281410232186317, + "learning_rate": 2.4488622888690785e-05, + "loss": 8.32078221719712e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00008, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.1868448704481125, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0016654160572215915, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00167, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.5, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.0011122338473796844, + "learning_rate": 2.406442693028651e-05, + "loss": 2.4154323909897357e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0008903060806915164, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.0764218788826838e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.004497055895626545, + "learning_rate": 2.3645380340187508e-05, + "loss": 3.050938539672643e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.004634706303477287, + "learning_rate": 2.3437809889624914e-05, + "loss": 4.90621714561712e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.59, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.00251275347545743, + "learning_rate": 2.3231552870624487e-05, + "loss": 4.33354580309242e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.004285548347979784, + "learning_rate": 2.3026617866382657e-05, + "loss": 5.757250255555846e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.08757233619689941, + "learning_rate": 2.2823013405081507e-05, + "loss": 2.351473449380137e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.00247983168810606, + "learning_rate": 2.2620747959533722e-05, + "loss": 4.2569590732455254e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00004, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.0006376878009177744, + "learning_rate": 2.2419829946830123e-05, + "loss": 1.580168100190349e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.010959293693304062, + "learning_rate": 2.2220267727989325e-05, + "loss": 9.101699106395245e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00009, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.0010561353992670774, + "learning_rate": 2.202206960760984e-05, + "loss": 1.6030653569032438e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.025496836751699448, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00019739707931876183, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0002, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.00516202999278903, + "learning_rate": 2.1629798596457056e-05, + "loss": 5.949653859715909e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0005809378926642239, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.5215588973660488e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.36, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0006219783099368215, + "learning_rate": 2.124308220868431e-05, + "loss": 1.3315910109668039e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00001, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0050822533667087555, + "learning_rate": 2.105182715082638e-05, + "loss": 2.660456084413454e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.81, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.004483669530600309, + "learning_rate": 2.0861984815011552e-05, + "loss": 4.43336321040988e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00004, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.002029991941526532, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.8533231670735404e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.0010850686812773347, + "learning_rate": 2.0486569850851317e-05, + "loss": 1.5834324585739523e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00002, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.22020931541919708, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00125799176748842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00126, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.003032378386706114, + "learning_rate": 2.011689980574966e-05, + "loss": 1.8148441085941158e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.01703350804746151, + "learning_rate": 1.993423839463052e-05, + "loss": 2.1378953533712775e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0006738354568369687, + "learning_rate": 1.975303621298445e-05, + "loss": 1.730798976495862e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.0016640514368191361, + "learning_rate": 1.957330080137385e-05, + "loss": 1.8207503671874292e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.05258834734559059, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00014502773410640657, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0004455884627532214, + "learning_rate": 1.9218260145006073e-05, + "loss": 1.1770534911192954e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.06, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.13378198444843292, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0007207246962934732, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.017833322286605835, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00016917653556447476, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00017, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.1, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.0006367648602463305, + "learning_rate": 1.869688492349885e-05, + "loss": 1.2300321031943895e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.00034314877120777965, + "learning_rate": 1.85261050441233e-05, + "loss": 1.0386990652477834e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 190216 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.11333048343658447, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.00011449151497799903, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00011, + "step": 417, + "tokens/total": 12617168, + "tokens/train_per_sec_per_gpu": 35.18, + "tokens/trainable": 190708 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.002743076765909791, + "learning_rate": 1.8189105812005714e-05, + "loss": 1.5843726941966452e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 418, + "tokens/total": 12647680, + "tokens/train_per_sec_per_gpu": 30.93, + "tokens/trainable": 191126 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.0011538882972672582, + "learning_rate": 1.802290048317732e-05, + "loss": 1.4711402400280349e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 419, + "tokens/total": 12677952, + "tokens/train_per_sec_per_gpu": 30.43, + "tokens/trainable": 191540 + }, + { + "epoch": 1.640625, + "grad_norm": 0.004467409569770098, + "learning_rate": 1.785823392239424e-05, + "loss": 4.823958806809969e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 420, + "tokens/total": 12708416, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.0012163098435848951, + "learning_rate": 1.7695112982104225e-05, + "loss": 1.1310569789202418e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00001, + "step": 421, + "tokens/total": 12738640, + "tokens/train_per_sec_per_gpu": 31.42, + "tokens/trainable": 192463 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.05586014315485954, + "learning_rate": 1.7533544450435433e-05, + "loss": 0.0002712146961130202, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00027, + "step": 422, + "tokens/total": 12769232, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 192918 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.06670165061950684, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.0004127066058572382, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00041, + "step": 423, + "tokens/total": 12799648, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 193425 + }, + { + "epoch": 1.65625, + "grad_norm": 0.019891072064638138, + "learning_rate": 1.721509144218405e-05, + "loss": 0.00023232161765918136, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00023, + "step": 424, + "tokens/total": 12829920, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 193898 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.006559237837791443, + "learning_rate": 1.705822021773101e-05, + "loss": 6.114803545642644e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 425, + "tokens/total": 12860112, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 194315 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.005621429067105055, + "learning_rate": 1.69029279056068e-05, + "loss": 6.088989175623283e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 426, + "tokens/total": 12890320, + "tokens/train_per_sec_per_gpu": 35.06, + "tokens/trainable": 194796 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.0004855124861933291, + "learning_rate": 1.6749220968158415e-05, + "loss": 1.0911244316957891e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00001, + "step": 427, + "tokens/total": 12920656, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 195262 + }, + { + "epoch": 1.671875, + "grad_norm": 0.0006959444144740701, + "learning_rate": 1.659710580175893e-05, + "loss": 1.4198772987583652e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 428, + "tokens/total": 12951040, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 195680 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.00048281854833476245, + "learning_rate": 1.644658873654133e-05, + "loss": 1.0396102879894897e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00001, + "step": 429, + "tokens/total": 12981232, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 196145 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.007472805213183165, + "learning_rate": 1.629767603613508e-05, + "loss": 5.188406430534087e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 430, + "tokens/total": 13011616, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 196626 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.008452537469565868, + "learning_rate": 1.615037389740547e-05, + "loss": 2.078613033518195e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00002, + "step": 431, + "tokens/total": 13042208, + "tokens/train_per_sec_per_gpu": 31.22, + "tokens/trainable": 197067 + }, + { + "epoch": 1.6875, + "grad_norm": 0.0006947954534552991, + "learning_rate": 1.600468845019576e-05, + "loss": 1.0801179087138735e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00001, + "step": 432, + "tokens/total": 13072800, + "tokens/train_per_sec_per_gpu": 38.63, + "tokens/trainable": 197565 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.0020175843965262175, + "learning_rate": 1.5860625757072092e-05, + "loss": 1.564873855386395e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 433, + "tokens/total": 13103328, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 198015 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.001340522081591189, + "learning_rate": 1.571819181307116e-05, + "loss": 1.7318390746368095e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 434, + "tokens/total": 13133472, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 198442 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.08985895663499832, + "learning_rate": 1.557739254545075e-05, + "loss": 0.0004795463755726814, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 435, + "tokens/total": 13164064, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 198914 + }, + { + "epoch": 1.703125, + "grad_norm": 0.0005816713673993945, + "learning_rate": 1.543823381344311e-05, + "loss": 1.203996089316206e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00001, + "step": 436, + "tokens/total": 13194464, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 199355 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.2994021773338318, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.0055721537210047245, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00559, + "step": 437, + "tokens/total": 13224992, + "tokens/train_per_sec_per_gpu": 32.98, + "tokens/trainable": 199830 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.0006708212895318866, + "learning_rate": 1.5164861051607254e-05, + "loss": 9.834290722210426e-06, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00001, + "step": 438, + "tokens/total": 13255296, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 200338 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.004102411679923534, + "learning_rate": 1.5030658397935521e-05, + "loss": 2.766325997072272e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00003, + "step": 439, + "tokens/total": 13285344, + "tokens/train_per_sec_per_gpu": 32.91, + "tokens/trainable": 200785 + }, + { + "epoch": 1.71875, + "grad_norm": 0.0009204484522342682, + "learning_rate": 1.4898119031716104e-05, + "loss": 1.4549179468303919e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00001, + "step": 440, + "tokens/total": 13315632, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 201267 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.0014299805043265224, + "learning_rate": 1.476724846845306e-05, + "loss": 2.365603722864762e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 441, + "tokens/total": 13346048, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 201734 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.01532546803355217, + "learning_rate": 1.463805215420471e-05, + "loss": 9.99852636596188e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 442, + "tokens/total": 13376304, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 202174 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.18507054448127747, + "learning_rate": 1.451053546535705e-05, + "loss": 0.000987955485470593, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00099, + "step": 443, + "tokens/total": 13406720, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 202612 + }, + { + "epoch": 1.734375, + "grad_norm": 0.028212063014507294, + "learning_rate": 1.438470370840001e-05, + "loss": 0.00025927621754817665, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00026, + "step": 444, + "tokens/total": 13436464, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 203020 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.0004877297324128449, + "learning_rate": 1.4260562119706606e-05, + "loss": 1.2777243682648987e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 445, + "tokens/total": 13466672, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 203479 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.02534927800297737, + "learning_rate": 1.413811586531508e-05, + "loss": 3.3006788726197556e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 446, + "tokens/total": 13496736, + "tokens/train_per_sec_per_gpu": 35.38, + "tokens/trainable": 203946 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.0006437217816710472, + "learning_rate": 1.4017370040713884e-05, + "loss": 1.1569853086257353e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 447, + "tokens/total": 13526928, + "tokens/train_per_sec_per_gpu": 37.38, + "tokens/trainable": 204382 + }, + { + "epoch": 1.75, + "grad_norm": 0.0014389768475666642, + "learning_rate": 1.3898329670629645e-05, + "loss": 2.1063802705612034e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 448, + "tokens/total": 13557392, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 204830 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.016985442489385605, + "learning_rate": 1.3780999708818058e-05, + "loss": 0.00012392934877425432, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00012, + "step": 449, + "tokens/total": 13587776, + "tokens/train_per_sec_per_gpu": 33.74, + "tokens/trainable": 205300 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.008965734392404556, + "learning_rate": 1.3665385037857758e-05, + "loss": 6.039683285052888e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 450, + "tokens/total": 13618112, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 205726 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.5729678273200989, + "learning_rate": 1.3551490468947126e-05, + "loss": 0.00580303929746151, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00582, + "step": 451, + "tokens/total": 13648512, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 206163 + }, + { + "epoch": 1.765625, + "grad_norm": 0.0277901329100132, + "learning_rate": 1.3439320741704075e-05, + "loss": 9.134741412708536e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 452, + "tokens/total": 13678704, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 206619 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.3788454532623291, + "learning_rate": 1.3328880523968808e-05, + "loss": 0.002095964504405856, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0021, + "step": 453, + "tokens/total": 13709232, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 207101 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.00038695961120538414, + "learning_rate": 1.3220174411609587e-05, + "loss": 9.780245818546973e-06, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00001, + "step": 454, + "tokens/total": 13739600, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 207546 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.01514297816902399, + "learning_rate": 1.3113206928331471e-05, + "loss": 7.800674939062446e-05, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 455, + "tokens/total": 13769936, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 207989 + }, + { + "epoch": 1.78125, + "grad_norm": 0.0018013569060713053, + "learning_rate": 1.300798252548806e-05, + "loss": 2.1191644918872043e-05, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00002, + "step": 456, + "tokens/total": 13800240, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 208402 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.0006610782584175467, + "learning_rate": 1.2904505581896265e-05, + "loss": 1.3429066711978521e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 457, + "tokens/total": 13830512, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 208904 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.0005749509437009692, + "learning_rate": 1.2802780403654082e-05, + "loss": 1.3295277312863618e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00001, + "step": 458, + "tokens/total": 13860832, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 209382 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.00635139923542738, + "learning_rate": 1.2702811223961408e-05, + "loss": 5.09147321281489e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 459, + "tokens/total": 13890992, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 209833 + }, + { + "epoch": 1.796875, + "grad_norm": 0.013369702734053135, + "learning_rate": 1.2604602202943861e-05, + "loss": 0.00011205296323169023, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 460, + "tokens/total": 13921424, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 210285 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.016762293875217438, + "learning_rate": 1.2508157427479686e-05, + "loss": 0.00019885516667272896, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.0002, + "step": 461, + "tokens/total": 13951504, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 210770 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.006627500057220459, + "learning_rate": 1.2413480911029655e-05, + "loss": 2.5290853955084458e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00003, + "step": 462, + "tokens/total": 13979728, + "tokens/train_per_sec_per_gpu": 30.5, + "tokens/trainable": 211194 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.003884925739839673, + "learning_rate": 1.2320576593470082e-05, + "loss": 2.766420948319137e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00003, + "step": 463, + "tokens/total": 14009936, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 211656 + }, + { + "epoch": 1.8125, + "grad_norm": 0.000377490563550964, + "learning_rate": 1.2229448340928828e-05, + "loss": 9.118146408582106e-06, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00001, + "step": 464, + "tokens/total": 14039872, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 212086 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.01634475402534008, + "learning_rate": 1.2140099945624458e-05, + "loss": 7.731329969828948e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 465, + "tokens/total": 14070096, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 212571 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.5773619413375854, + "learning_rate": 1.205253512570841e-05, + "loss": 0.008904650807380676, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00894, + "step": 466, + "tokens/total": 14100192, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 213010 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.005223860964179039, + "learning_rate": 1.1966757525110255e-05, + "loss": 3.063467738684267e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 467, + "tokens/total": 14130448, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 213449 + }, + { + "epoch": 1.828125, + "grad_norm": 0.00045610740198753774, + "learning_rate": 1.1882770713386095e-05, + "loss": 1.0150557500310242e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00001, + "step": 468, + "tokens/total": 14160480, + "tokens/train_per_sec_per_gpu": 35.84, + "tokens/trainable": 213897 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.005935850087553263, + "learning_rate": 1.180057818556998e-05, + "loss": 4.396100121084601e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00004, + "step": 469, + "tokens/total": 14191152, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 214365 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.0010427028173580766, + "learning_rate": 1.1720183362028494e-05, + "loss": 2.14513493119739e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 470, + "tokens/total": 14219392, + "tokens/train_per_sec_per_gpu": 35.86, + "tokens/trainable": 214799 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.006076959893107414, + "learning_rate": 1.1641589588318387e-05, + "loss": 1.3421818948700093e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00001, + "step": 471, + "tokens/total": 14249920, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 215291 + }, + { + "epoch": 1.84375, + "grad_norm": 0.0019434966379776597, + "learning_rate": 1.1564800135047418e-05, + "loss": 3.379978079465218e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00003, + "step": 472, + "tokens/total": 14280048, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 215740 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.01917141303420067, + "learning_rate": 1.148981819773816e-05, + "loss": 8.30673161544837e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 473, + "tokens/total": 14310432, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 216193 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.027403976768255234, + "learning_rate": 1.1416646896695086e-05, + "loss": 5.644366319756955e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00006, + "step": 474, + "tokens/total": 14340496, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 216633 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.0030669430270791054, + "learning_rate": 1.1345289276874717e-05, + "loss": 3.4152639273088425e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 475, + "tokens/total": 14370704, + "tokens/train_per_sec_per_gpu": 30.62, + "tokens/trainable": 217039 + }, + { + "epoch": 1.859375, + "grad_norm": 0.0022093369625508785, + "learning_rate": 1.1275748307758873e-05, + "loss": 3.003427991643548e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 476, + "tokens/total": 14401120, + "tokens/train_per_sec_per_gpu": 37.85, + "tokens/trainable": 217529 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.007151913829147816, + "learning_rate": 1.1208026883231147e-05, + "loss": 4.787252328242175e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00005, + "step": 477, + "tokens/total": 14431344, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 217987 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.005195859353989363, + "learning_rate": 1.1142127821456433e-05, + "loss": 6.279069930315018e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 478, + "tokens/total": 14461952, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 218460 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.0024305693805217743, + "learning_rate": 1.1078053864763674e-05, + "loss": 4.7390385589096695e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00005, + "step": 479, + "tokens/total": 14491968, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 218908 + }, + { + "epoch": 1.875, + "grad_norm": 0.0007113535539247096, + "learning_rate": 1.1015807679531756e-05, + "loss": 1.3227703675511293e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00001, + "step": 480, + "tokens/total": 14522528, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 219349 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.857288412958648e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-480/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..16470d79c556f350f8eb6fed9179426818a41f86 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:004757923662cc6573db04eb1edf3bcc6f5152f20b0e388191af7ad75b8ba2f5 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..ec4388e93b2aa17f2947adba0a154186002f125c --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:65916b4beb82a197b99e16dd2db1ea96ed374c4fc8a805d7a27ee6ac4b30a97a +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..93906269157a5485207c9bb4b3b757d5ea700afa --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:75e98dcceb68f1b6ee34a96307625206231cf5a0e0e4e16d8a9eafccc33ea3fe +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..22325b3b3d08e0697784fe21b8c321485a51f7b5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4c9a27ec19f62c50830c1143627b3abecff0b0cc --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/tokens_state.json @@ -0,0 +1 @@ +{"total": 15491744, "trainable": 234076} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0872c9d98d4cd022d3b9d9e265ff76dc7aec9228 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/trainer_state.json @@ -0,0 +1,7202 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 512, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.3597196340560913, + "learning_rate": 9.53619478457953e-05, + "loss": 0.0036215470172464848, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00363, + "step": 97, + "tokens/total": 2936752, + "tokens/train_per_sec_per_gpu": 31.62, + "tokens/trainable": 44255 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.7050648927688599, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015324651263654232, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01544, + "step": 98, + "tokens/total": 2967072, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 44738 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.5716716051101685, + "learning_rate": 9.51018809682839e-05, + "loss": 0.013399260118603706, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01349, + "step": 99, + "tokens/total": 2997664, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 45201 + }, + { + "epoch": 0.390625, + "grad_norm": 0.18090857565402985, + "learning_rate": 9.49693416020645e-05, + "loss": 0.0020832926966249943, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00209, + "step": 100, + "tokens/total": 3028032, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 45645 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.5402258038520813, + "learning_rate": 9.483513894839276e-05, + "loss": 0.011402487754821777, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01147, + "step": 101, + "tokens/total": 3058272, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 46088 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.220508873462677, + "learning_rate": 9.469927859198888e-05, + "loss": 0.026607759296894073, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02696, + "step": 102, + "tokens/total": 3088848, + "tokens/train_per_sec_per_gpu": 32.58, + "tokens/trainable": 46544 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.488730788230896, + "learning_rate": 9.456176618655689e-05, + "loss": 0.007838367484509945, + "memory/device_reserved (GiB)": 35.99, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00787, + "step": 103, + "tokens/total": 3119424, + "tokens/train_per_sec_per_gpu": 31.96, + "tokens/trainable": 47011 + }, + { + "epoch": 0.40625, + "grad_norm": 1.0092511177062988, + "learning_rate": 9.442260745454927e-05, + "loss": 0.024805881083011627, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02512, + "step": 104, + "tokens/total": 3149696, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 47500 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.1705712527036667, + "learning_rate": 9.428180818692884e-05, + "loss": 0.002884519286453724, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00289, + "step": 105, + "tokens/total": 3180064, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 47993 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.539692759513855, + "learning_rate": 9.413937424292791e-05, + "loss": 0.028211303055286407, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02861, + "step": 106, + "tokens/total": 3208320, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 48444 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.576267421245575, + "learning_rate": 9.399531154980424e-05, + "loss": 0.026665428653359413, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02702, + "step": 107, + "tokens/total": 3238416, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 48886 + }, + { + "epoch": 0.421875, + "grad_norm": 0.19187162816524506, + "learning_rate": 9.384962610259455e-05, + "loss": 0.007470840122550726, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0075, + "step": 108, + "tokens/total": 3268784, + "tokens/train_per_sec_per_gpu": 37.78, + "tokens/trainable": 49357 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.30193838477134705, + "learning_rate": 9.370232396386494e-05, + "loss": 0.012815583497285843, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0129, + "step": 109, + "tokens/total": 3298912, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 49824 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.24559386074543, + "learning_rate": 9.355341126345868e-05, + "loss": 0.010728488676249981, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01079, + "step": 110, + "tokens/total": 3329232, + "tokens/train_per_sec_per_gpu": 34.26, + "tokens/trainable": 50264 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3296513259410858, + "learning_rate": 9.340289419824107e-05, + "loss": 0.01623804122209549, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01637, + "step": 111, + "tokens/total": 3359440, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 50708 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4614265263080597, + "learning_rate": 9.325077903184159e-05, + "loss": 0.021365080028772354, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02159, + "step": 112, + "tokens/total": 3389472, + "tokens/train_per_sec_per_gpu": 31.61, + "tokens/trainable": 51154 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.28496578335762024, + "learning_rate": 9.30970720943932e-05, + "loss": 0.013247976079583168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01334, + "step": 113, + "tokens/total": 3419920, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 51610 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.5088275074958801, + "learning_rate": 9.2941779782269e-05, + "loss": 0.022269586101174355, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.02252, + "step": 114, + "tokens/total": 3450176, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 52014 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.19142404198646545, + "learning_rate": 9.278490855781596e-05, + "loss": 0.006432589143514633, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00645, + "step": 115, + "tokens/total": 3480544, + "tokens/train_per_sec_per_gpu": 36.32, + "tokens/trainable": 52503 + }, + { + "epoch": 0.453125, + "grad_norm": 1.382026195526123, + "learning_rate": 9.262646494908604e-05, + "loss": 0.012845169752836227, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01293, + "step": 116, + "tokens/total": 3510848, + "tokens/train_per_sec_per_gpu": 38.75, + "tokens/trainable": 53005 + }, + { + "epoch": 0.45703125, + "grad_norm": 0.16703951358795166, + "learning_rate": 9.246645554956457e-05, + "loss": 0.0037160126958042383, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00372, + "step": 117, + "tokens/total": 3541296, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 53464 + }, + { + "epoch": 0.4609375, + "grad_norm": 0.615856945514679, + "learning_rate": 9.230488701789578e-05, + "loss": 0.014962448738515377, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01507, + "step": 118, + "tokens/total": 3571936, + "tokens/train_per_sec_per_gpu": 32.88, + "tokens/trainable": 53920 + }, + { + "epoch": 0.46484375, + "grad_norm": 0.24910427629947662, + "learning_rate": 9.214176607760577e-05, + "loss": 0.004819548688828945, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00483, + "step": 119, + "tokens/total": 3602000, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 54360 + }, + { + "epoch": 0.46875, + "grad_norm": 0.9213384389877319, + "learning_rate": 9.197709951682268e-05, + "loss": 0.008400149643421173, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00844, + "step": 120, + "tokens/total": 3632288, + "tokens/train_per_sec_per_gpu": 35.21, + "tokens/trainable": 54836 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.1228351816534996, + "learning_rate": 9.181089418799428e-05, + "loss": 0.0019936836324632168, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.002, + "step": 121, + "tokens/total": 3662608, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 55264 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.18029853701591492, + "learning_rate": 9.164315700760271e-05, + "loss": 0.004047749564051628, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00406, + "step": 122, + "tokens/total": 3692864, + "tokens/train_per_sec_per_gpu": 30.55, + "tokens/trainable": 55697 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.4368670582771301, + "learning_rate": 9.147389495587671e-05, + "loss": 0.00867566466331482, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00871, + "step": 123, + "tokens/total": 3722864, + "tokens/train_per_sec_per_gpu": 33.49, + "tokens/trainable": 56141 + }, + { + "epoch": 0.484375, + "grad_norm": 0.3315965235233307, + "learning_rate": 9.130311507650116e-05, + "loss": 0.011374368332326412, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01144, + "step": 124, + "tokens/total": 3753056, + "tokens/train_per_sec_per_gpu": 33.2, + "tokens/trainable": 56586 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.2834460735321045, + "learning_rate": 9.113082447632394e-05, + "loss": 0.005234894342720509, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00525, + "step": 125, + "tokens/total": 3783312, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 57078 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.5132001638412476, + "learning_rate": 9.09570303250602e-05, + "loss": 0.02469645068049431, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.025, + "step": 126, + "tokens/total": 3813712, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 57524 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.9969733953475952, + "learning_rate": 9.078173985499394e-05, + "loss": 0.040580250322818756, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.04141, + "step": 127, + "tokens/total": 3844336, + "tokens/train_per_sec_per_gpu": 34.85, + "tokens/trainable": 58002 + }, + { + "epoch": 0.5, + "grad_norm": 0.4804457724094391, + "learning_rate": 9.060496036067713e-05, + "loss": 0.007047833874821663, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00707, + "step": 128, + "tokens/total": 3872688, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 58451 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.23698025941848755, + "learning_rate": 9.042669919862615e-05, + "loss": 0.00366212404333055, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00367, + "step": 129, + "tokens/total": 3902640, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 58893 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.44587212800979614, + "learning_rate": 9.024696378701557e-05, + "loss": 0.010764295235276222, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01082, + "step": 130, + "tokens/total": 3933104, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 59353 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3212004005908966, + "learning_rate": 9.006576160536948e-05, + "loss": 0.013213565573096275, + "memory/device_reserved (GiB)": 34.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0133, + "step": 131, + "tokens/total": 3963184, + "tokens/train_per_sec_per_gpu": 32.29, + "tokens/trainable": 59789 + }, + { + "epoch": 0.515625, + "grad_norm": 0.14710628986358643, + "learning_rate": 8.988310019425035e-05, + "loss": 0.005224708467721939, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00524, + "step": 132, + "tokens/total": 3993648, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 60263 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.5616672039031982, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01082855835556984, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01089, + "step": 133, + "tokens/total": 4024000, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 60756 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.28706833720207214, + "learning_rate": 8.951343014914869e-05, + "loss": 0.02170894667506218, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.02195, + "step": 134, + "tokens/total": 4052480, + "tokens/train_per_sec_per_gpu": 31.85, + "tokens/trainable": 61201 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.1448316127061844, + "learning_rate": 8.932643689864568e-05, + "loss": 0.0064529795199632645, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00647, + "step": 135, + "tokens/total": 4082864, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 61681 + }, + { + "epoch": 0.53125, + "grad_norm": 0.2812711000442505, + "learning_rate": 8.913801518498845e-05, + "loss": 0.01243551354855299, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.01251, + "step": 136, + "tokens/total": 4113040, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 62156 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.37281039357185364, + "learning_rate": 8.894817284917364e-05, + "loss": 0.009842045605182648, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00989, + "step": 137, + "tokens/total": 4143344, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 62646 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.15259796380996704, + "learning_rate": 8.875691779131569e-05, + "loss": 0.007893089205026627, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00792, + "step": 138, + "tokens/total": 4173648, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 63097 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.2609741985797882, + "learning_rate": 8.856425797031829e-05, + "loss": 0.009887185879051685, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00994, + "step": 139, + "tokens/total": 4204080, + "tokens/train_per_sec_per_gpu": 37.62, + "tokens/trainable": 63585 + }, + { + "epoch": 0.546875, + "grad_norm": 0.3938979506492615, + "learning_rate": 8.837020140354295e-05, + "loss": 0.020112136378884315, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.02032, + "step": 140, + "tokens/total": 4232400, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 64002 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2883419096469879, + "learning_rate": 8.817475616647554e-05, + "loss": 0.020921282470226288, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02114, + "step": 141, + "tokens/total": 4262624, + "tokens/train_per_sec_per_gpu": 35.28, + "tokens/trainable": 64459 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.22204594314098358, + "learning_rate": 8.797793039239017e-05, + "loss": 0.007596657145768404, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00763, + "step": 142, + "tokens/total": 4293168, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 64897 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.3463776409626007, + "learning_rate": 8.777973227201069e-05, + "loss": 0.008941063657402992, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00898, + "step": 143, + "tokens/total": 4323168, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 65376 + }, + { + "epoch": 0.5625, + "grad_norm": 0.07992643117904663, + "learning_rate": 8.758017005316988e-05, + "loss": 0.002066058572381735, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00207, + "step": 144, + "tokens/total": 4353264, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 65842 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.4542894959449768, + "learning_rate": 8.737925204046629e-05, + "loss": 0.006295363884419203, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00632, + "step": 145, + "tokens/total": 4383504, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 66282 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.6165645122528076, + "learning_rate": 8.717698659491851e-05, + "loss": 0.021687351167201996, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.02192, + "step": 146, + "tokens/total": 4413776, + "tokens/train_per_sec_per_gpu": 31.67, + "tokens/trainable": 66733 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.44622310996055603, + "learning_rate": 8.697338213361735e-05, + "loss": 0.010542265139520168, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0106, + "step": 147, + "tokens/total": 4444032, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 67147 + }, + { + "epoch": 0.578125, + "grad_norm": 0.5972599387168884, + "learning_rate": 8.676844712937552e-05, + "loss": 0.015109038911759853, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01522, + "step": 148, + "tokens/total": 4474384, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 67630 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.16330066323280334, + "learning_rate": 8.656219011037509e-05, + "loss": 0.003038810333237052, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00304, + "step": 149, + "tokens/total": 4504800, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 68074 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.1966242492198944, + "learning_rate": 8.63546196598125e-05, + "loss": 0.00474123377352953, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00475, + "step": 150, + "tokens/total": 4534896, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 68542 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.2519090175628662, + "learning_rate": 8.614574441554145e-05, + "loss": 0.005299043841660023, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00531, + "step": 151, + "tokens/total": 4565008, + "tokens/train_per_sec_per_gpu": 30.34, + "tokens/trainable": 69008 + }, + { + "epoch": 0.59375, + "grad_norm": 0.45198729634284973, + "learning_rate": 8.593557306971349e-05, + "loss": 0.00804843008518219, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00808, + "step": 152, + "tokens/total": 4595520, + "tokens/train_per_sec_per_gpu": 35.52, + "tokens/trainable": 69462 + }, + { + "epoch": 0.59765625, + "grad_norm": 1.9722840785980225, + "learning_rate": 8.572411436841618e-05, + "loss": 0.015857091173529625, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01598, + "step": 153, + "tokens/total": 4625952, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 69925 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.1563708782196045, + "learning_rate": 8.551137711130922e-05, + "loss": 0.003014163114130497, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00302, + "step": 154, + "tokens/total": 4656272, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 70392 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.3952409327030182, + "learning_rate": 8.529737015125824e-05, + "loss": 0.020546773448586464, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02076, + "step": 155, + "tokens/total": 4684592, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 70826 + }, + { + "epoch": 0.609375, + "grad_norm": 0.29893431067466736, + "learning_rate": 8.508210239396639e-05, + "loss": 0.006672243122011423, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00669, + "step": 156, + "tokens/total": 4715136, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 71284 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.12646262347698212, + "learning_rate": 8.486558279760375e-05, + "loss": 0.001321017974987626, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00132, + "step": 157, + "tokens/total": 4745296, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 71746 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.3663196563720703, + "learning_rate": 8.464782037243449e-05, + "loss": 0.00867768656462431, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00872, + "step": 158, + "tokens/total": 4775696, + "tokens/train_per_sec_per_gpu": 32.33, + "tokens/trainable": 72182 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.32793474197387695, + "learning_rate": 8.442882418044202e-05, + "loss": 0.012621743604540825, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0127, + "step": 159, + "tokens/total": 4805776, + "tokens/train_per_sec_per_gpu": 30.69, + "tokens/trainable": 72601 + }, + { + "epoch": 0.625, + "grad_norm": 0.502180278301239, + "learning_rate": 8.420860333495179e-05, + "loss": 0.02071015164256096, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02093, + "step": 160, + "tokens/total": 4836224, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 73058 + }, + { + "epoch": 0.62890625, + "grad_norm": 3.3360722064971924, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010355101898312569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.01041, + "step": 161, + "tokens/total": 4866464, + "tokens/train_per_sec_per_gpu": 30.11, + "tokens/trainable": 73471 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.42214301228523254, + "learning_rate": 8.376452439121266e-05, + "loss": 0.010006662458181381, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01006, + "step": 162, + "tokens/total": 4897040, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 73920 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.5886772871017456, + "learning_rate": 8.354068477290124e-05, + "loss": 0.015542788431048393, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01566, + "step": 163, + "tokens/total": 4927360, + "tokens/train_per_sec_per_gpu": 35.47, + "tokens/trainable": 74376 + }, + { + "epoch": 0.640625, + "grad_norm": 0.5640157461166382, + "learning_rate": 8.331565746019807e-05, + "loss": 0.019208863377571106, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01939, + "step": 164, + "tokens/total": 4957952, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 74827 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.09329716116189957, + "learning_rate": 8.308945181740812e-05, + "loss": 0.001608099672012031, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00161, + "step": 165, + "tokens/total": 4988288, + "tokens/train_per_sec_per_gpu": 31.79, + "tokens/trainable": 75259 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.4663255512714386, + "learning_rate": 8.286207725787153e-05, + "loss": 0.003368590958416462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00337, + "step": 166, + "tokens/total": 5018592, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 75700 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.09389737993478775, + "learning_rate": 8.263354324357182e-05, + "loss": 0.004297872073948383, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00431, + "step": 167, + "tokens/total": 5049264, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 76161 + }, + { + "epoch": 0.65625, + "grad_norm": 0.2795652151107788, + "learning_rate": 8.240385928474219e-05, + "loss": 0.009957612492144108, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01001, + "step": 168, + "tokens/total": 5079776, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 76620 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.6108075976371765, + "learning_rate": 8.217303493946967e-05, + "loss": 0.004623654298484325, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00463, + "step": 169, + "tokens/total": 5110112, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 77044 + }, + { + "epoch": 0.6640625, + "grad_norm": 2.5686914920806885, + "learning_rate": 8.194107981329746e-05, + "loss": 0.014518939889967442, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01462, + "step": 170, + "tokens/total": 5140512, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 77493 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.29714030027389526, + "learning_rate": 8.170800355882518e-05, + "loss": 0.010281096212565899, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.01033, + "step": 171, + "tokens/total": 5170768, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 77953 + }, + { + "epoch": 0.671875, + "grad_norm": 0.24662797152996063, + "learning_rate": 8.147381587530713e-05, + "loss": 0.005124978721141815, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00514, + "step": 172, + "tokens/total": 5201200, + "tokens/train_per_sec_per_gpu": 29.9, + "tokens/trainable": 78407 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.19063295423984528, + "learning_rate": 8.123852650824877e-05, + "loss": 0.005921770352870226, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00594, + "step": 173, + "tokens/total": 5231632, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 78846 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.5131625533103943, + "learning_rate": 8.100214524900103e-05, + "loss": 0.01540191750973463, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.01552, + "step": 174, + "tokens/total": 5261360, + "tokens/train_per_sec_per_gpu": 30.17, + "tokens/trainable": 79263 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.11886871606111526, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0013660730328410864, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00137, + "step": 175, + "tokens/total": 5291648, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 79733 + }, + { + "epoch": 0.6875, + "grad_norm": 0.39346420764923096, + "learning_rate": 8.052614644612253e-05, + "loss": 0.007847018539905548, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00788, + "step": 176, + "tokens/total": 5322160, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 80211 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.05455538257956505, + "learning_rate": 8.028654871074489e-05, + "loss": 0.0007541990489698946, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00075, + "step": 177, + "tokens/total": 5352496, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 80658 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.3038773536682129, + "learning_rate": 8.004589869885986e-05, + "loss": 0.006179198157042265, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0062, + "step": 178, + "tokens/total": 5382704, + "tokens/train_per_sec_per_gpu": 34.89, + "tokens/trainable": 81126 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.08510788530111313, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0012173540890216827, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00122, + "step": 179, + "tokens/total": 5412704, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 81573 + }, + { + "epoch": 0.703125, + "grad_norm": 0.3349142074584961, + "learning_rate": 7.95614819466576e-05, + "loss": 0.00507398834452033, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00509, + "step": 180, + "tokens/total": 5442880, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 82046 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.3200117349624634, + "learning_rate": 7.931773536489872e-05, + "loss": 0.007542713545262814, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00757, + "step": 181, + "tokens/total": 5473136, + "tokens/train_per_sec_per_gpu": 38.92, + "tokens/trainable": 82563 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.13203248381614685, + "learning_rate": 7.907297682291035e-05, + "loss": 0.0025186133570969105, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00252, + "step": 182, + "tokens/total": 5503488, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 83051 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.3643631637096405, + "learning_rate": 7.882721650609442e-05, + "loss": 0.012744799256324768, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01283, + "step": 183, + "tokens/total": 5533680, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 83498 + }, + { + "epoch": 0.71875, + "grad_norm": 0.012706859968602657, + "learning_rate": 7.85804646415409e-05, + "loss": 0.00015147411613725126, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00015, + "step": 184, + "tokens/total": 5564160, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 84004 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.14483962953090668, + "learning_rate": 7.833273149760207e-05, + "loss": 0.0014616945991292596, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00146, + "step": 185, + "tokens/total": 5594368, + "tokens/train_per_sec_per_gpu": 35.63, + "tokens/trainable": 84488 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.3152420222759247, + "learning_rate": 7.808402738346527e-05, + "loss": 0.005956828128546476, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00597, + "step": 186, + "tokens/total": 5624736, + "tokens/train_per_sec_per_gpu": 33.45, + "tokens/trainable": 84949 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.15444505214691162, + "learning_rate": 7.783436264872382e-05, + "loss": 0.0022030179388821125, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00221, + "step": 187, + "tokens/total": 5655024, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 85397 + }, + { + "epoch": 0.734375, + "grad_norm": 1.1173254251480103, + "learning_rate": 7.758374768294647e-05, + "loss": 0.018108580261468887, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01827, + "step": 188, + "tokens/total": 5685280, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 85886 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.15292231738567352, + "learning_rate": 7.733219291524489e-05, + "loss": 0.0033938095439225435, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0034, + "step": 189, + "tokens/total": 5715568, + "tokens/train_per_sec_per_gpu": 36.24, + "tokens/trainable": 86393 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.3460272550582886, + "learning_rate": 7.707970881383977e-05, + "loss": 0.003943222109228373, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00395, + "step": 190, + "tokens/total": 5746160, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 86864 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.17786552011966705, + "learning_rate": 7.682630588562518e-05, + "loss": 0.002941778162494302, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00295, + "step": 191, + "tokens/total": 5776736, + "tokens/train_per_sec_per_gpu": 36.08, + "tokens/trainable": 87312 + }, + { + "epoch": 0.75, + "grad_norm": 0.05713065341114998, + "learning_rate": 7.657199467573129e-05, + "loss": 0.0006791478954255581, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00068, + "step": 192, + "tokens/total": 5807168, + "tokens/train_per_sec_per_gpu": 30.28, + "tokens/trainable": 87734 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.21285323798656464, + "learning_rate": 7.631678576708561e-05, + "loss": 0.005839701741933823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00586, + "step": 193, + "tokens/total": 5837600, + "tokens/train_per_sec_per_gpu": 35.12, + "tokens/trainable": 88217 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.046665649861097336, + "learning_rate": 7.606068977997255e-05, + "loss": 0.0009576653246767819, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00096, + "step": 194, + "tokens/total": 5867920, + "tokens/train_per_sec_per_gpu": 36.11, + "tokens/trainable": 88697 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.12541399896144867, + "learning_rate": 7.580371737159148e-05, + "loss": 0.002426251769065857, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00243, + "step": 195, + "tokens/total": 5896144, + "tokens/train_per_sec_per_gpu": 30.41, + "tokens/trainable": 89130 + }, + { + "epoch": 0.765625, + "grad_norm": 0.06126544624567032, + "learning_rate": 7.554587923561324e-05, + "loss": 0.000532104168087244, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00053, + "step": 196, + "tokens/total": 5926544, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 89592 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.11023399233818054, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0028196191415190697, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00282, + "step": 197, + "tokens/total": 5954864, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 90040 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.16828177869319916, + "learning_rate": 7.502764873523431e-05, + "loss": 0.004482921212911606, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00449, + "step": 198, + "tokens/total": 5985120, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 90501 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.19191870093345642, + "learning_rate": 7.476727793652011e-05, + "loss": 0.002102922648191452, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00211, + "step": 199, + "tokens/total": 6015872, + "tokens/train_per_sec_per_gpu": 36.51, + "tokens/trainable": 90980 + }, + { + "epoch": 0.78125, + "grad_norm": 0.0961354449391365, + "learning_rate": 7.450608454068415e-05, + "loss": 0.0013840183382853866, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00138, + "step": 200, + "tokens/total": 6046128, + "tokens/train_per_sec_per_gpu": 39.36, + "tokens/trainable": 91464 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.03724909946322441, + "learning_rate": 7.424407941704987e-05, + "loss": 0.00039239737088792026, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00039, + "step": 201, + "tokens/total": 6076128, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 91910 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.47981590032577515, + "learning_rate": 7.398127346871986e-05, + "loss": 0.008898628875613213, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00894, + "step": 202, + "tokens/total": 6106480, + "tokens/train_per_sec_per_gpu": 37.79, + "tokens/trainable": 92383 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.25382256507873535, + "learning_rate": 7.371767763212238e-05, + "loss": 0.02967994660139084, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03012, + "step": 203, + "tokens/total": 6136560, + "tokens/train_per_sec_per_gpu": 36.6, + "tokens/trainable": 92840 + }, + { + "epoch": 0.796875, + "grad_norm": 0.11337354779243469, + "learning_rate": 7.345330287655617e-05, + "loss": 0.0005227055517025292, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00052, + "step": 204, + "tokens/total": 6167104, + "tokens/train_per_sec_per_gpu": 34.02, + "tokens/trainable": 93321 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.08194643259048462, + "learning_rate": 7.31881602037339e-05, + "loss": 0.0015989269595593214, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.0016, + "step": 205, + "tokens/total": 6197296, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 93754 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.42092061042785645, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005056938622146845, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00507, + "step": 206, + "tokens/total": 6227520, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 94233 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.28918036818504333, + "learning_rate": 7.265561527249383e-05, + "loss": 0.00922542903572321, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00927, + "step": 207, + "tokens/total": 6257680, + "tokens/train_per_sec_per_gpu": 36.57, + "tokens/trainable": 94731 + }, + { + "epoch": 0.8125, + "grad_norm": 0.052897270768880844, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008248636149801314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00083, + "step": 208, + "tokens/total": 6287968, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 95216 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.053124021738767624, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0011008362052962184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0011, + "step": 209, + "tokens/total": 6318608, + "tokens/train_per_sec_per_gpu": 29.28, + "tokens/trainable": 95617 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.3910926580429077, + "learning_rate": 7.185131535190975e-05, + "loss": 0.0025537661276757717, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00256, + "step": 210, + "tokens/total": 6348864, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 96074 + }, + { + "epoch": 0.82421875, + "grad_norm": 0.1602534055709839, + "learning_rate": 7.158179796885005e-05, + "loss": 0.0042940047569572926, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0043, + "step": 211, + "tokens/total": 6378880, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 96503 + }, + { + "epoch": 0.828125, + "grad_norm": 0.22192947566509247, + "learning_rate": 7.131159054949273e-05, + "loss": 0.002052969066426158, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00206, + "step": 212, + "tokens/total": 6409392, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 96928 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.17284627258777618, + "learning_rate": 7.104070433827139e-05, + "loss": 0.0034506958909332752, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00346, + "step": 213, + "tokens/total": 6439824, + "tokens/train_per_sec_per_gpu": 32.77, + "tokens/trainable": 97359 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.2631441056728363, + "learning_rate": 7.076915060786705e-05, + "loss": 0.0022546069230884314, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00226, + "step": 214, + "tokens/total": 6470096, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 97856 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.24480876326560974, + "learning_rate": 7.049694065873882e-05, + "loss": 0.005642582196742296, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00566, + "step": 215, + "tokens/total": 6500336, + "tokens/train_per_sec_per_gpu": 37.45, + "tokens/trainable": 98306 + }, + { + "epoch": 0.84375, + "grad_norm": 0.01338098756968975, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0002527001779526472, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00025, + "step": 216, + "tokens/total": 6530688, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 98779 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.01576339825987816, + "learning_rate": 6.99505974422157e-05, + "loss": 0.00029742170590907335, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0003, + "step": 217, + "tokens/total": 6561040, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 99264 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.14224660396575928, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0019305831519886851, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00193, + "step": 218, + "tokens/total": 6589392, + "tokens/train_per_sec_per_gpu": 39.69, + "tokens/trainable": 99729 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.13956739008426666, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0020216472912579775, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00202, + "step": 219, + "tokens/total": 6619824, + "tokens/train_per_sec_per_gpu": 31.73, + "tokens/trainable": 100162 + }, + { + "epoch": 0.859375, + "grad_norm": 0.2409037947654724, + "learning_rate": 6.912644503343682e-05, + "loss": 0.0016943392110988498, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0017, + "step": 220, + "tokens/total": 6650224, + "tokens/train_per_sec_per_gpu": 36.42, + "tokens/trainable": 100619 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.004593817982822657, + "learning_rate": 6.885053657779273e-05, + "loss": 0.00010279623529640958, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 221, + "tokens/total": 6680384, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 101079 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.3027840256690979, + "learning_rate": 6.857405174478604e-05, + "loss": 0.0038326543290168047, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00384, + "step": 222, + "tokens/total": 6710896, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 101545 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.06953064352273941, + "learning_rate": 6.82970020400792e-05, + "loss": 0.0005575605318881571, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00056, + "step": 223, + "tokens/total": 6741280, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 101986 + }, + { + "epoch": 0.875, + "grad_norm": 0.14961817860603333, + "learning_rate": 6.801939899284132e-05, + "loss": 0.0019154315814375877, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00192, + "step": 224, + "tokens/total": 6771488, + "tokens/train_per_sec_per_gpu": 35.51, + "tokens/trainable": 102434 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.1315905600786209, + "learning_rate": 6.774125415526827e-05, + "loss": 0.0011086631566286087, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00111, + "step": 225, + "tokens/total": 6801744, + "tokens/train_per_sec_per_gpu": 29.24, + "tokens/trainable": 102842 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7103981375694275, + "learning_rate": 6.746257910210214e-05, + "loss": 0.007341520860791206, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00737, + "step": 226, + "tokens/total": 6832432, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 103341 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.014314249157905579, + "learning_rate": 6.718338543014937e-05, + "loss": 0.0001612441410543397, + "memory/device_reserved (GiB)": 35.71, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00016, + "step": 227, + "tokens/total": 6862448, + "tokens/train_per_sec_per_gpu": 40.93, + "tokens/trainable": 103843 + }, + { + "epoch": 0.890625, + "grad_norm": 0.00822363793849945, + "learning_rate": 6.69036847577983e-05, + "loss": 9.871406655292958e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0001, + "step": 228, + "tokens/total": 6892832, + "tokens/train_per_sec_per_gpu": 36.35, + "tokens/trainable": 104306 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.08351919800043106, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0004943685489706695, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00049, + "step": 229, + "tokens/total": 6922832, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 104771 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.004186064004898071, + "learning_rate": 6.63428089904618e-05, + "loss": 6.949146336410195e-05, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 230, + "tokens/total": 6953104, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 105239 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.008335944265127182, + "learning_rate": 6.60616572358065e-05, + "loss": 0.00010860025940928608, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00011, + "step": 231, + "tokens/total": 6983440, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 105702 + }, + { + "epoch": 0.90625, + "grad_norm": 0.11052291095256805, + "learning_rate": 6.578004516044172e-05, + "loss": 0.0004777174908667803, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00048, + "step": 232, + "tokens/total": 7013840, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 106178 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.28182464838027954, + "learning_rate": 6.549798448339548e-05, + "loss": 0.0048217386938631535, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00483, + "step": 233, + "tokens/total": 7044240, + "tokens/train_per_sec_per_gpu": 38.1, + "tokens/trainable": 106682 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.19313767552375793, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0006764436257071793, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00068, + "step": 234, + "tokens/total": 7074224, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 107120 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.005185638088732958, + "learning_rate": 6.493256429322259e-05, + "loss": 6.307615694822744e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00006, + "step": 235, + "tokens/total": 7104672, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 107560 + }, + { + "epoch": 0.921875, + "grad_norm": 0.07209689170122147, + "learning_rate": 6.464922830953799e-05, + "loss": 0.0004264797898940742, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00043, + "step": 236, + "tokens/total": 7134784, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 108017 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.005625654011964798, + "learning_rate": 6.436549078207688e-05, + "loss": 4.9979436880676076e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 237, + "tokens/total": 7164960, + "tokens/train_per_sec_per_gpu": 33.47, + "tokens/trainable": 108445 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.008250541053712368, + "learning_rate": 6.408136351831592e-05, + "loss": 6.751110777258873e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00007, + "step": 238, + "tokens/total": 7195344, + "tokens/train_per_sec_per_gpu": 29.25, + "tokens/trainable": 108865 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.5217323899269104, + "learning_rate": 6.379685834195036e-05, + "loss": 0.0005912262131460011, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00059, + "step": 239, + "tokens/total": 7225680, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 109315 + }, + { + "epoch": 0.9375, + "grad_norm": 0.011697136797010899, + "learning_rate": 6.351198709240186e-05, + "loss": 9.851530194282532e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0001, + "step": 240, + "tokens/total": 7256112, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 109786 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.008918526582419872, + "learning_rate": 6.32267616243259e-05, + "loss": 6.70133886160329e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00007, + "step": 241, + "tokens/total": 7286384, + "tokens/train_per_sec_per_gpu": 38.51, + "tokens/trainable": 110245 + }, + { + "epoch": 0.9453125, + "grad_norm": 1.0020146369934082, + "learning_rate": 6.294119380711849e-05, + "loss": 0.013925625942647457, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01402, + "step": 242, + "tokens/total": 7316688, + "tokens/train_per_sec_per_gpu": 29.29, + "tokens/trainable": 110666 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.00132578460033983, + "learning_rate": 6.265529552442209e-05, + "loss": 1.891679858090356e-05, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 243, + "tokens/total": 7346992, + "tokens/train_per_sec_per_gpu": 38.21, + "tokens/trainable": 111153 + }, + { + "epoch": 0.953125, + "grad_norm": 0.898801326751709, + "learning_rate": 6.236907867363127e-05, + "loss": 0.012029297649860382, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0121, + "step": 244, + "tokens/total": 7377632, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 111610 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.17281004786491394, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001876274705864489, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00188, + "step": 245, + "tokens/total": 7407872, + "tokens/train_per_sec_per_gpu": 34.28, + "tokens/trainable": 112101 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.8393605947494507, + "learning_rate": 6.179573692313344e-05, + "loss": 0.02536478079855442, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02569, + "step": 246, + "tokens/total": 7438448, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 112568 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.1928865760564804, + "learning_rate": 6.150863588251694e-05, + "loss": 0.001368399360217154, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00137, + "step": 247, + "tokens/total": 7466768, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 113022 + }, + { + "epoch": 0.96875, + "grad_norm": 0.11117105931043625, + "learning_rate": 6.122126399099419e-05, + "loss": 0.0006689627189189196, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00067, + "step": 248, + "tokens/total": 7497360, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 113463 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.041484661400318146, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.0006179807242006063, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00062, + "step": 249, + "tokens/total": 7527504, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 113907 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0462549552321434, + "learning_rate": 6.064575550087316e-05, + "loss": 0.000625512795522809, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00063, + "step": 250, + "tokens/total": 7555808, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 114347 + }, + { + "epoch": 0.98046875, + "grad_norm": 0.23392240703105927, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.004232748411595821, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00424, + "step": 251, + "tokens/total": 7586320, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 114829 + }, + { + "epoch": 0.984375, + "grad_norm": 0.3223143517971039, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.007172638550400734, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0072, + "step": 252, + "tokens/total": 7616576, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 115277 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.1359666883945465, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.003720771986991167, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00373, + "step": 253, + "tokens/total": 7647152, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 115698 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.10392674058675766, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.0012953465338796377, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0013, + "step": 254, + "tokens/total": 7677584, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 116132 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.1488286703824997, + "learning_rate": 5.920308275189541e-05, + "loss": 0.0010544590186327696, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00106, + "step": 255, + "tokens/total": 7708016, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 116590 + }, + { + "epoch": 1.0, + "grad_norm": 0.025846293196082115, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.00045689160469919443, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00046, + "step": 256, + "tokens/total": 7738608, + "tokens/train_per_sec_per_gpu": 32.19, + "tokens/trainable": 117038 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.011676807887852192, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.00018332478066440672, + "memory/device_reserved (GiB)": 36.13, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00018, + "step": 257, + "tokens/total": 7768944, + "tokens/train_per_sec_per_gpu": 32.51, + "tokens/trainable": 117494 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.026530586183071136, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0004955548793077469, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0005, + "step": 258, + "tokens/total": 7799232, + "tokens/train_per_sec_per_gpu": 30.63, + "tokens/trainable": 117924 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.03306467458605766, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0007393938140012324, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00074, + "step": 259, + "tokens/total": 7829568, + "tokens/train_per_sec_per_gpu": 31.76, + "tokens/trainable": 118372 + }, + { + "epoch": 1.015625, + "grad_norm": 0.01490688230842352, + "learning_rate": 5.77560376810977e-05, + "loss": 0.00033902074210345745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00034, + "step": 260, + "tokens/total": 7860000, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 118815 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.23672910034656525, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.003403924172744155, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00341, + "step": 261, + "tokens/total": 7890464, + "tokens/train_per_sec_per_gpu": 33.82, + "tokens/trainable": 119262 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.11774802953004837, + "learning_rate": 5.717633247526522e-05, + "loss": 0.002715344773605466, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00272, + "step": 262, + "tokens/total": 7920944, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 119771 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.028089547529816628, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0006553111597895622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00066, + "step": 263, + "tokens/total": 7951296, + "tokens/train_per_sec_per_gpu": 33.9, + "tokens/trainable": 120268 + }, + { + "epoch": 1.03125, + "grad_norm": 0.04741915315389633, + "learning_rate": 5.659626500889066e-05, + "loss": 0.0009464840404689312, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00095, + "step": 264, + "tokens/total": 7981856, + "tokens/train_per_sec_per_gpu": 37.13, + "tokens/trainable": 120745 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.06473700702190399, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.0010454395087435842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00105, + "step": 265, + "tokens/total": 8012192, + "tokens/train_per_sec_per_gpu": 38.37, + "tokens/trainable": 121203 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.1152043268084526, + "learning_rate": 5.601593183686955e-05, + "loss": 0.0010041436180472374, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.001, + "step": 266, + "tokens/total": 8042736, + "tokens/train_per_sec_per_gpu": 31.29, + "tokens/trainable": 121669 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.1012289822101593, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0007158135995268822, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00072, + "step": 267, + "tokens/total": 8073024, + "tokens/train_per_sec_per_gpu": 40.14, + "tokens/trainable": 122182 + }, + { + "epoch": 1.046875, + "grad_norm": 0.014074633829295635, + "learning_rate": 5.543542955832538e-05, + "loss": 0.00022851164976600558, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00023, + "step": 268, + "tokens/total": 8103280, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 122644 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.022633297368884087, + "learning_rate": 5.514514519946986e-05, + "loss": 0.00056973792379722, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00057, + "step": 269, + "tokens/total": 8133440, + "tokens/train_per_sec_per_gpu": 33.16, + "tokens/trainable": 123104 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.028963575139641762, + "learning_rate": 5.485485480053015e-05, + "loss": 0.00030712541774846613, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00031, + "step": 270, + "tokens/total": 8163920, + "tokens/train_per_sec_per_gpu": 36.27, + "tokens/trainable": 123594 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.05075100436806679, + "learning_rate": 5.4564570441674645e-05, + "loss": 0.0004000376502517611, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0004, + "step": 271, + "tokens/total": 8194032, + "tokens/train_per_sec_per_gpu": 37.71, + "tokens/trainable": 124076 + }, + { + "epoch": 1.0625, + "grad_norm": 0.014762450009584427, + "learning_rate": 5.42743042028204e-05, + "loss": 0.0001975473714992404, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0002, + "step": 272, + "tokens/total": 8224320, + "tokens/train_per_sec_per_gpu": 34.82, + "tokens/trainable": 124546 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.3096662759780884, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0021046444308012724, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00211, + "step": 273, + "tokens/total": 8254368, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 124976 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.14023423194885254, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0011167863849550486, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00112, + "step": 274, + "tokens/total": 8284944, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 125437 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.006373860873281956, + "learning_rate": 5.340373499110935e-05, + "loss": 0.00010782821482280269, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00011, + "step": 275, + "tokens/total": 8315584, + "tokens/train_per_sec_per_gpu": 31.54, + "tokens/trainable": 125888 + }, + { + "epoch": 1.078125, + "grad_norm": 0.44675078988075256, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0035920855589210987, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0036, + "step": 276, + "tokens/total": 8345952, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 126362 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.0032541481778025627, + "learning_rate": 5.282366752473479e-05, + "loss": 5.1655006245709956e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 277, + "tokens/total": 8376416, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 126827 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.14338454604148865, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.0026056424248963594, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00261, + "step": 278, + "tokens/total": 8406640, + "tokens/train_per_sec_per_gpu": 38.13, + "tokens/trainable": 127330 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.20478253066539764, + "learning_rate": 5.224396231890232e-05, + "loss": 0.0012996959267184138, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0013, + "step": 279, + "tokens/total": 8436784, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 127797 + }, + { + "epoch": 1.09375, + "grad_norm": 0.0009973018895834684, + "learning_rate": 5.195427572104522e-05, + "loss": 2.559608401497826e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 280, + "tokens/total": 8467152, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 128246 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.001558113843202591, + "learning_rate": 5.166471586820751e-05, + "loss": 3.741440741578117e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 281, + "tokens/total": 8497424, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 128728 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.01588490605354309, + "learning_rate": 5.1375294810156615e-05, + "loss": 8.873045590007678e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00009, + "step": 282, + "tokens/total": 8527616, + "tokens/train_per_sec_per_gpu": 35.14, + "tokens/trainable": 129202 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.06619437038898468, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0010551117593422532, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00106, + "step": 283, + "tokens/total": 8557984, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 129652 + }, + { + "epoch": 1.109375, + "grad_norm": 0.1938449889421463, + "learning_rate": 5.079691724810461e-05, + "loss": 0.0008955710800364614, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0009, + "step": 284, + "tokens/total": 8588384, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 130099 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.16278043389320374, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.0006343181594274938, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00063, + "step": 285, + "tokens/total": 8618768, + "tokens/train_per_sec_per_gpu": 30.56, + "tokens/trainable": 130542 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.0029784520156681538, + "learning_rate": 5.021923930849237e-05, + "loss": 4.526478369371034e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 286, + "tokens/total": 8648976, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 130999 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.06447792798280716, + "learning_rate": 4.99306927511967e-05, + "loss": 5.6983706599567086e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 287, + "tokens/total": 8679152, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 131448 + }, + { + "epoch": 1.125, + "grad_norm": 0.00280313054099679, + "learning_rate": 4.964235714846775e-05, + "loss": 4.409470420796424e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00004, + "step": 288, + "tokens/total": 8709504, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 131922 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.0028473951388150454, + "learning_rate": 4.9354244499126866e-05, + "loss": 3.198662307113409e-05, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 289, + "tokens/total": 8739680, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 132396 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.03604506328701973, + "learning_rate": 4.90663667927174e-05, + "loss": 0.00023753194545861334, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00024, + "step": 290, + "tokens/total": 8770112, + "tokens/train_per_sec_per_gpu": 36.46, + "tokens/trainable": 132845 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.2052794098854065, + "learning_rate": 4.877873600900581e-05, + "loss": 0.0011388716520741582, + "memory/device_reserved (GiB)": 35.57, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00114, + "step": 291, + "tokens/total": 8800368, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 133346 + }, + { + "epoch": 1.140625, + "grad_norm": 0.12911155819892883, + "learning_rate": 4.849136411748306e-05, + "loss": 0.00046830251812934875, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00047, + "step": 292, + "tokens/total": 8830688, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 133778 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.0032441529911011457, + "learning_rate": 4.8204263076866574e-05, + "loss": 4.348478250904009e-05, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 293, + "tokens/total": 8861008, + "tokens/train_per_sec_per_gpu": 36.39, + "tokens/trainable": 134250 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.1916627436876297, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0007836729055270553, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00078, + "step": 294, + "tokens/total": 8891760, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 134682 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.11497358977794647, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0007789967930875719, + "memory/device_reserved (GiB)": 35.59, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00078, + "step": 295, + "tokens/total": 8922192, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 135164 + }, + { + "epoch": 1.15625, + "grad_norm": 0.07835118472576141, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.00046658708015456796, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00047, + "step": 296, + "tokens/total": 8952496, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 135598 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.43025892972946167, + "learning_rate": 4.705880619288153e-05, + "loss": 0.011139214970171452, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0112, + "step": 297, + "tokens/total": 8982768, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 136037 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.0018146105576306581, + "learning_rate": 4.677323837567412e-05, + "loss": 2.296476304763928e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 298, + "tokens/total": 9013248, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 136494 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.013126488775014877, + "learning_rate": 4.6488012907598146e-05, + "loss": 9.889354987535626e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0001, + "step": 299, + "tokens/total": 9043632, + "tokens/train_per_sec_per_gpu": 35.41, + "tokens/trainable": 136952 + }, + { + "epoch": 1.171875, + "grad_norm": 0.04675251618027687, + "learning_rate": 4.620314165804964e-05, + "loss": 0.0003341895353514701, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00033, + "step": 300, + "tokens/total": 9073936, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 137411 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.02300048992037773, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0001898434857139364, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00019, + "step": 301, + "tokens/total": 9104416, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 137906 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.0014230191009119153, + "learning_rate": 4.5634509217923135e-05, + "loss": 2.900467188737821e-05, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 302, + "tokens/total": 9134688, + "tokens/train_per_sec_per_gpu": 34.33, + "tokens/trainable": 138382 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.01370098628103733, + "learning_rate": 4.535077169046201e-05, + "loss": 0.0001739814761094749, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00017, + "step": 303, + "tokens/total": 9165088, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 138841 + }, + { + "epoch": 1.1875, + "grad_norm": 0.12571607530117035, + "learning_rate": 4.506743570677743e-05, + "loss": 0.00018397449457552284, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00018, + "step": 304, + "tokens/total": 9195584, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 139268 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2337377369403839, + "learning_rate": 4.478451305763618e-05, + "loss": 0.003163608256727457, + "memory/device_reserved (GiB)": 35.67, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00317, + "step": 305, + "tokens/total": 9225760, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 139735 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.029297636821866035, + "learning_rate": 4.450201551660454e-05, + "loss": 0.00035399867920204997, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00035, + "step": 306, + "tokens/total": 9256352, + "tokens/train_per_sec_per_gpu": 37.06, + "tokens/trainable": 140225 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.07336489856243134, + "learning_rate": 4.4219954839558276e-05, + "loss": 0.0006284696282818913, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00063, + "step": 307, + "tokens/total": 9284256, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 140663 + }, + { + "epoch": 1.203125, + "grad_norm": 0.0036061827559024096, + "learning_rate": 4.393834276419352e-05, + "loss": 3.499729427858256e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 308, + "tokens/total": 9314768, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 141114 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.03760785236954689, + "learning_rate": 4.36571910095382e-05, + "loss": 0.0002682818449102342, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00027, + "step": 309, + "tokens/total": 9344976, + "tokens/train_per_sec_per_gpu": 28.93, + "tokens/trainable": 141529 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.005482436623424292, + "learning_rate": 4.337651127546448e-05, + "loss": 7.971160812303424e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 310, + "tokens/total": 9375280, + "tokens/train_per_sec_per_gpu": 34.27, + "tokens/trainable": 141963 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.006238086149096489, + "learning_rate": 4.3096315242201736e-05, + "loss": 4.3831034417962655e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 311, + "tokens/total": 9405536, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 142427 + }, + { + "epoch": 1.21875, + "grad_norm": 0.00502822594717145, + "learning_rate": 4.2816614569850635e-05, + "loss": 5.458533996716142e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 312, + "tokens/total": 9435696, + "tokens/train_per_sec_per_gpu": 38.17, + "tokens/trainable": 142893 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.17115379869937897, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.0009321674006059766, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00093, + "step": 313, + "tokens/total": 9466080, + "tokens/train_per_sec_per_gpu": 36.45, + "tokens/trainable": 143367 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.009989157319068909, + "learning_rate": 4.225874584473174e-05, + "loss": 0.00014035131607670337, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00014, + "step": 314, + "tokens/total": 9496400, + "tokens/train_per_sec_per_gpu": 32.97, + "tokens/trainable": 143823 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.006546014454215765, + "learning_rate": 4.19806010071587e-05, + "loss": 7.929251296445727e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00008, + "step": 315, + "tokens/total": 9526880, + "tokens/train_per_sec_per_gpu": 34.63, + "tokens/trainable": 144305 + }, + { + "epoch": 1.234375, + "grad_norm": 0.017240718007087708, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0001234864175785333, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 316, + "tokens/total": 9557280, + "tokens/train_per_sec_per_gpu": 35.53, + "tokens/trainable": 144762 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.011387556791305542, + "learning_rate": 4.142594825521398e-05, + "loss": 8.356192847713828e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00008, + "step": 317, + "tokens/total": 9587712, + "tokens/train_per_sec_per_gpu": 37.5, + "tokens/trainable": 145254 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.007485110778361559, + "learning_rate": 4.114946342220728e-05, + "loss": 8.404521213378757e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00008, + "step": 318, + "tokens/total": 9618064, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 145713 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.24612121284008026, + "learning_rate": 4.087355496656321e-05, + "loss": 0.0037109816912561655, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00372, + "step": 319, + "tokens/total": 9648336, + "tokens/train_per_sec_per_gpu": 36.28, + "tokens/trainable": 146200 + }, + { + "epoch": 1.25, + "grad_norm": 0.00244643772020936, + "learning_rate": 4.05982343699588e-05, + "loss": 4.710642315330915e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 320, + "tokens/total": 9678688, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 146668 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.018739959225058556, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.00018197213648818433, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 321, + "tokens/total": 9708848, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 147114 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.22725771367549896, + "learning_rate": 4.004940255778431e-05, + "loss": 0.002088680863380432, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00209, + "step": 322, + "tokens/total": 9739360, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 147564 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.000548983458429575, + "learning_rate": 3.977591418134619e-05, + "loss": 1.6327385310432874e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 323, + "tokens/total": 9769776, + "tokens/train_per_sec_per_gpu": 34.31, + "tokens/trainable": 148013 + }, + { + "epoch": 1.265625, + "grad_norm": 0.00761087192222476, + "learning_rate": 3.95030593412612e-05, + "loss": 5.408083234215155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00005, + "step": 324, + "tokens/total": 9800096, + "tokens/train_per_sec_per_gpu": 38.23, + "tokens/trainable": 148470 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.008723607286810875, + "learning_rate": 3.923084939213296e-05, + "loss": 8.67105700308457e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00009, + "step": 325, + "tokens/total": 9830304, + "tokens/train_per_sec_per_gpu": 36.62, + "tokens/trainable": 148926 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.006354155018925667, + "learning_rate": 3.895929566172861e-05, + "loss": 5.4418724175775424e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 326, + "tokens/total": 9860608, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 149420 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.7132415771484375, + "learning_rate": 3.868840945050728e-05, + "loss": 0.008354030549526215, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00839, + "step": 327, + "tokens/total": 9890560, + "tokens/train_per_sec_per_gpu": 39.57, + "tokens/trainable": 149873 + }, + { + "epoch": 1.28125, + "grad_norm": 0.0944104865193367, + "learning_rate": 3.841820203114995e-05, + "loss": 0.0006982197519391775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0007, + "step": 328, + "tokens/total": 9920880, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 150347 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.0028970104176551104, + "learning_rate": 3.814868464809027e-05, + "loss": 3.3767173590604216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00003, + "step": 329, + "tokens/total": 9951280, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 150822 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.009064619429409504, + "learning_rate": 3.787986851704667e-05, + "loss": 7.040683703962713e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 330, + "tokens/total": 9981600, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 151266 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.05380578711628914, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00042552349623292685, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00043, + "step": 331, + "tokens/total": 10012016, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 151734 + }, + { + "epoch": 1.296875, + "grad_norm": 0.00875640194863081, + "learning_rate": 3.734438472750619e-05, + "loss": 6.885632319608703e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00007, + "step": 332, + "tokens/total": 10042448, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 152184 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.0009642363293096423, + "learning_rate": 3.707773935267552e-05, + "loss": 2.1829437173437327e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00002, + "step": 333, + "tokens/total": 10073024, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 152623 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.0004874366568401456, + "learning_rate": 3.68118397962661e-05, + "loss": 1.1361282304278575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00001, + "step": 334, + "tokens/total": 10103504, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 153040 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.31265145540237427, + "learning_rate": 3.654669712344384e-05, + "loss": 0.009118539281189442, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.31, + "memory/max_allocated (GiB)": 33.31, + "ppl": 1.00916, + "step": 335, + "tokens/total": 10131632, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 153463 + }, + { + "epoch": 1.3125, + "grad_norm": 0.0011828432325273752, + "learning_rate": 3.628232236787763e-05, + "loss": 2.4649223632877693e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00002, + "step": 336, + "tokens/total": 10161824, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 153908 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.008298320695757866, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.00010702587314881384, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 337, + "tokens/total": 10192448, + "tokens/train_per_sec_per_gpu": 33.84, + "tokens/trainable": 154387 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.0037000810261815786, + "learning_rate": 3.575592058295017e-05, + "loss": 3.555983494152315e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00004, + "step": 338, + "tokens/total": 10222704, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 154836 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.30193620920181274, + "learning_rate": 3.549391545931585e-05, + "loss": 0.0002394289622316137, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00024, + "step": 339, + "tokens/total": 10253168, + "tokens/train_per_sec_per_gpu": 29.94, + "tokens/trainable": 155261 + }, + { + "epoch": 1.328125, + "grad_norm": 0.013805638998746872, + "learning_rate": 3.5232722063479914e-05, + "loss": 9.105106437345967e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00009, + "step": 340, + "tokens/total": 10283552, + "tokens/train_per_sec_per_gpu": 31.24, + "tokens/trainable": 155674 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.26545384526252747, + "learning_rate": 3.49723512647657e-05, + "loss": 0.011088686995208263, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01115, + "step": 341, + "tokens/total": 10313872, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 156164 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.00922760646790266, + "learning_rate": 3.471281389826491e-05, + "loss": 9.105133358389139e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 342, + "tokens/total": 10344256, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 156611 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.14466270804405212, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0014950309414416552, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.0015, + "step": 343, + "tokens/total": 10374544, + "tokens/train_per_sec_per_gpu": 33.8, + "tokens/trainable": 157073 + }, + { + "epoch": 1.34375, + "grad_norm": 0.10482759773731232, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0009619007469154894, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00096, + "step": 344, + "tokens/total": 10405040, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 157523 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.20004107058048248, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.002495028544217348, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0025, + "step": 345, + "tokens/total": 10435424, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 157964 + }, + { + "epoch": 1.3515625, + "grad_norm": 8.192997932434082, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.006594266276806593, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00662, + "step": 346, + "tokens/total": 10465760, + "tokens/train_per_sec_per_gpu": 30.54, + "tokens/trainable": 158390 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.005605250597000122, + "learning_rate": 3.342800532426873e-05, + "loss": 6.323108391370624e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00006, + "step": 347, + "tokens/total": 10495904, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 158848 + }, + { + "epoch": 1.359375, + "grad_norm": 0.003419809276238084, + "learning_rate": 3.317369411437484e-05, + "loss": 5.6915632740128785e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00006, + "step": 348, + "tokens/total": 10526624, + "tokens/train_per_sec_per_gpu": 28.85, + "tokens/trainable": 159270 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.18140868842601776, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0037076647859066725, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00371, + "step": 349, + "tokens/total": 10556752, + "tokens/train_per_sec_per_gpu": 32.67, + "tokens/trainable": 159720 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.002614055061712861, + "learning_rate": 3.266780708475511e-05, + "loss": 3.247601125622168e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00003, + "step": 350, + "tokens/total": 10587264, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 160187 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.002343076979741454, + "learning_rate": 3.241625231705354e-05, + "loss": 4.7991154133342206e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 351, + "tokens/total": 10617504, + "tokens/train_per_sec_per_gpu": 32.81, + "tokens/trainable": 160645 + }, + { + "epoch": 1.375, + "grad_norm": 0.03407059609889984, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0003904080658685416, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00039, + "step": 352, + "tokens/total": 10647760, + "tokens/train_per_sec_per_gpu": 34.71, + "tokens/trainable": 161102 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.012719057500362396, + "learning_rate": 3.191597261653475e-05, + "loss": 0.00012954325939062983, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 353, + "tokens/total": 10678080, + "tokens/train_per_sec_per_gpu": 29.06, + "tokens/trainable": 161555 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.005359884351491928, + "learning_rate": 3.166726850239794e-05, + "loss": 8.441988029517233e-05, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 354, + "tokens/total": 10708544, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 162020 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.056369855999946594, + "learning_rate": 3.141953535845912e-05, + "loss": 0.00027662873617373407, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00028, + "step": 355, + "tokens/total": 10739024, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 162448 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0013983466196805239, + "learning_rate": 3.11727834939056e-05, + "loss": 3.224632018827833e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00003, + "step": 356, + "tokens/total": 10769024, + "tokens/train_per_sec_per_gpu": 28.48, + "tokens/trainable": 162863 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.06720387935638428, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0002829490404110402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00028, + "step": 357, + "tokens/total": 10799328, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 163318 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.01048748753964901, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.00019201112445443869, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00019, + "step": 358, + "tokens/total": 10829488, + "tokens/train_per_sec_per_gpu": 33.28, + "tokens/trainable": 163767 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.11898642778396606, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.0006578543689101934, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00066, + "step": 359, + "tokens/total": 10859936, + "tokens/train_per_sec_per_gpu": 32.13, + "tokens/trainable": 164208 + }, + { + "epoch": 1.40625, + "grad_norm": 0.09823726862668991, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.001951234764419496, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00195, + "step": 360, + "tokens/total": 10890400, + "tokens/train_per_sec_per_gpu": 34.39, + "tokens/trainable": 164655 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.37214013934135437, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.003241142490878701, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00325, + "step": 361, + "tokens/total": 10920624, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 165100 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.033950626850128174, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00024406128795817494, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00024, + "step": 362, + "tokens/total": 10951136, + "tokens/train_per_sec_per_gpu": 35.19, + "tokens/trainable": 165587 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.11754101514816284, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.0018770417664200068, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00188, + "step": 363, + "tokens/total": 10981344, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 166034 + }, + { + "epoch": 1.421875, + "grad_norm": 0.012853973545134068, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00021306828421074897, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 364, + "tokens/total": 11011552, + "tokens/train_per_sec_per_gpu": 34.3, + "tokens/trainable": 166485 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.09332777559757233, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00026301087928004563, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00026, + "step": 365, + "tokens/total": 11041872, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 166973 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.009343368001282215, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.0001053380110533908, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00011, + "step": 366, + "tokens/total": 11072192, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 167414 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.035727277398109436, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0005394808831624687, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00054, + "step": 367, + "tokens/total": 11102736, + "tokens/train_per_sec_per_gpu": 30.4, + "tokens/trainable": 167851 + }, + { + "epoch": 1.4375, + "grad_norm": 0.08836446702480316, + "learning_rate": 2.829199644117484e-05, + "loss": 0.000713829998858273, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00071, + "step": 368, + "tokens/total": 11133056, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 168294 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.006047180388122797, + "learning_rate": 2.8058920186702553e-05, + "loss": 8.545963646611199e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00009, + "step": 369, + "tokens/total": 11163568, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 168769 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.21181856095790863, + "learning_rate": 2.782696506053033e-05, + "loss": 0.0027023768052458763, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00271, + "step": 370, + "tokens/total": 11193936, + "tokens/train_per_sec_per_gpu": 37.04, + "tokens/trainable": 169270 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001994711346924305, + "learning_rate": 2.7596140715257824e-05, + "loss": 3.8951005990384147e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00004, + "step": 371, + "tokens/total": 11224480, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 169711 + }, + { + "epoch": 1.453125, + "grad_norm": 0.0069135697558522224, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.00011239978630328551, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00011, + "step": 372, + "tokens/total": 11254912, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 170185 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.3907967507839203, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0018560648895800114, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00186, + "step": 373, + "tokens/total": 11285104, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 170650 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.0009331480250693858, + "learning_rate": 2.691054818259188e-05, + "loss": 2.2479640392703004e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00002, + "step": 374, + "tokens/total": 11315600, + "tokens/train_per_sec_per_gpu": 32.22, + "tokens/trainable": 171126 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.14613144099712372, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0031683319248259068, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00317, + "step": 375, + "tokens/total": 11345776, + "tokens/train_per_sec_per_gpu": 35.85, + "tokens/trainable": 171600 + }, + { + "epoch": 1.46875, + "grad_norm": 0.00874971691519022, + "learning_rate": 2.645931522709877e-05, + "loss": 7.25445497664623e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00007, + "step": 376, + "tokens/total": 11376208, + "tokens/train_per_sec_per_gpu": 28.3, + "tokens/trainable": 172017 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.02506718970835209, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.00010988111171172932, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00011, + "step": 377, + "tokens/total": 11406512, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 172493 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.09279019385576248, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005218038568273187, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11436544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 172951 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.002128554042428732, + "learning_rate": 2.579139666504821e-05, + "loss": 3.309818930574693e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00003, + "step": 379, + "tokens/total": 11467008, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 173435 + }, + { + "epoch": 1.484375, + "grad_norm": 0.08014458417892456, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00045113274245522916, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00045, + "step": 380, + "tokens/total": 11497184, + "tokens/train_per_sec_per_gpu": 30.65, + "tokens/trainable": 173900 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.009996578097343445, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.00013630215835291892, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00014, + "step": 381, + "tokens/total": 11527600, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 174371 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.0028510303236544132, + "learning_rate": 2.5134417202396277e-05, + "loss": 2.7343612600816414e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 382, + "tokens/total": 11557808, + "tokens/train_per_sec_per_gpu": 32.95, + "tokens/trainable": 174826 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.0006682629464194179, + "learning_rate": 2.491789760603361e-05, + "loss": 9.333229172625579e-06, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 383, + "tokens/total": 11588352, + "tokens/train_per_sec_per_gpu": 33.1, + "tokens/trainable": 175276 + }, + { + "epoch": 1.5, + "grad_norm": 0.0072454060427844524, + "learning_rate": 2.4702629848741764e-05, + "loss": 8.043196430662647e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00008, + "step": 384, + "tokens/total": 11618640, + "tokens/train_per_sec_per_gpu": 33.88, + "tokens/trainable": 175720 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.011281410232186317, + "learning_rate": 2.4488622888690785e-05, + "loss": 8.32078221719712e-05, + "memory/device_reserved (GiB)": 35.89, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00008, + "step": 385, + "tokens/total": 11649072, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 176203 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.1868448704481125, + "learning_rate": 2.427588563158384e-05, + "loss": 0.0016654160572215915, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00167, + "step": 386, + "tokens/total": 11679552, + "tokens/train_per_sec_per_gpu": 34.5, + "tokens/trainable": 176651 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.0011122338473796844, + "learning_rate": 2.406442693028651e-05, + "loss": 2.4154323909897357e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00002, + "step": 387, + "tokens/total": 11709760, + "tokens/train_per_sec_per_gpu": 32.89, + "tokens/trainable": 177115 + }, + { + "epoch": 1.515625, + "grad_norm": 0.0008903060806915164, + "learning_rate": 2.3854255584458547e-05, + "loss": 2.0764218788826838e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 388, + "tokens/total": 11740288, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 177601 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.004497055895626545, + "learning_rate": 2.3645380340187508e-05, + "loss": 3.050938539672643e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 389, + "tokens/total": 11770592, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 178050 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.004634706303477287, + "learning_rate": 2.3437809889624914e-05, + "loss": 4.90621714561712e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 390, + "tokens/total": 11800752, + "tokens/train_per_sec_per_gpu": 36.59, + "tokens/trainable": 178531 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.00251275347545743, + "learning_rate": 2.3231552870624487e-05, + "loss": 4.33354580309242e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 391, + "tokens/total": 11831360, + "tokens/train_per_sec_per_gpu": 34.13, + "tokens/trainable": 179000 + }, + { + "epoch": 1.53125, + "grad_norm": 0.004285548347979784, + "learning_rate": 2.3026617866382657e-05, + "loss": 5.757250255555846e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00006, + "step": 392, + "tokens/total": 11861552, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 179457 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.08757233619689941, + "learning_rate": 2.2823013405081507e-05, + "loss": 2.351473449380137e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 393, + "tokens/total": 11891904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 179890 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.00247983168810606, + "learning_rate": 2.2620747959533722e-05, + "loss": 4.2569590732455254e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00004, + "step": 394, + "tokens/total": 11922208, + "tokens/train_per_sec_per_gpu": 33.99, + "tokens/trainable": 180341 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.0006376878009177744, + "learning_rate": 2.2419829946830123e-05, + "loss": 1.580168100190349e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 395, + "tokens/total": 11952672, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 180762 + }, + { + "epoch": 1.546875, + "grad_norm": 0.010959293693304062, + "learning_rate": 2.2220267727989325e-05, + "loss": 9.101699106395245e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00009, + "step": 396, + "tokens/total": 11983088, + "tokens/train_per_sec_per_gpu": 35.3, + "tokens/trainable": 181226 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.0010561353992670774, + "learning_rate": 2.202206960760984e-05, + "loss": 1.6030653569032438e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 397, + "tokens/total": 12013488, + "tokens/train_per_sec_per_gpu": 33.31, + "tokens/trainable": 181714 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.025496836751699448, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00019739707931876183, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0002, + "step": 398, + "tokens/total": 12043968, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 182142 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.00516202999278903, + "learning_rate": 2.1629798596457056e-05, + "loss": 5.949653859715909e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 399, + "tokens/total": 12074160, + "tokens/train_per_sec_per_gpu": 34.96, + "tokens/trainable": 182604 + }, + { + "epoch": 1.5625, + "grad_norm": 0.0005809378926642239, + "learning_rate": 2.1435742029681725e-05, + "loss": 1.5215588973660488e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00002, + "step": 400, + "tokens/total": 12104352, + "tokens/train_per_sec_per_gpu": 25.36, + "tokens/trainable": 183011 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.0006219783099368215, + "learning_rate": 2.124308220868431e-05, + "loss": 1.3315910109668039e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00001, + "step": 401, + "tokens/total": 12134240, + "tokens/train_per_sec_per_gpu": 31.9, + "tokens/trainable": 183459 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0050822533667087555, + "learning_rate": 2.105182715082638e-05, + "loss": 2.660456084413454e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 402, + "tokens/total": 12164480, + "tokens/train_per_sec_per_gpu": 38.81, + "tokens/trainable": 183959 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.004483669530600309, + "learning_rate": 2.0861984815011552e-05, + "loss": 4.43336321040988e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00004, + "step": 403, + "tokens/total": 12194640, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 184431 + }, + { + "epoch": 1.578125, + "grad_norm": 0.002029991941526532, + "learning_rate": 2.0673563101354323e-05, + "loss": 2.8533231670735404e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 404, + "tokens/total": 12224960, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 184867 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.0010850686812773347, + "learning_rate": 2.0486569850851317e-05, + "loss": 1.5834324585739523e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00002, + "step": 405, + "tokens/total": 12255008, + "tokens/train_per_sec_per_gpu": 29.46, + "tokens/trainable": 185300 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.22020931541919708, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.00125799176748842, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00126, + "step": 406, + "tokens/total": 12285248, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 185754 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.003032378386706114, + "learning_rate": 2.011689980574966e-05, + "loss": 1.8148441085941158e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 407, + "tokens/total": 12315552, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 186213 + }, + { + "epoch": 1.59375, + "grad_norm": 0.01703350804746151, + "learning_rate": 1.993423839463052e-05, + "loss": 2.1378953533712775e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 408, + "tokens/total": 12345904, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 186659 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.0006738354568369687, + "learning_rate": 1.975303621298445e-05, + "loss": 1.730798976495862e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00002, + "step": 409, + "tokens/total": 12376336, + "tokens/train_per_sec_per_gpu": 33.12, + "tokens/trainable": 187083 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.0016640514368191361, + "learning_rate": 1.957330080137385e-05, + "loss": 1.8207503671874292e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 410, + "tokens/total": 12406880, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 187530 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.05258834734559059, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00014502773410640657, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 411, + "tokens/total": 12437088, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 187972 + }, + { + "epoch": 1.609375, + "grad_norm": 0.0004455884627532214, + "learning_rate": 1.9218260145006073e-05, + "loss": 1.1770534911192954e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00001, + "step": 412, + "tokens/total": 12465440, + "tokens/train_per_sec_per_gpu": 40.06, + "tokens/trainable": 188417 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.13378198444843292, + "learning_rate": 1.904296967493982e-05, + "loss": 0.0007207246962934732, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 413, + "tokens/total": 12495968, + "tokens/train_per_sec_per_gpu": 31.66, + "tokens/trainable": 188868 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.017833322286605835, + "learning_rate": 1.8869175523676064e-05, + "loss": 0.00016917653556447476, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00017, + "step": 414, + "tokens/total": 12526128, + "tokens/train_per_sec_per_gpu": 37.1, + "tokens/trainable": 189337 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.0006367648602463305, + "learning_rate": 1.869688492349885e-05, + "loss": 1.2300321031943895e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 415, + "tokens/total": 12556256, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 189750 + }, + { + "epoch": 1.625, + "grad_norm": 0.00034314877120777965, + "learning_rate": 1.85261050441233e-05, + "loss": 1.0386990652477834e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 416, + "tokens/total": 12586736, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 190216 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.11333048343658447, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.00011449151497799903, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00011, + "step": 417, + "tokens/total": 12617168, + "tokens/train_per_sec_per_gpu": 35.18, + "tokens/trainable": 190708 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.002743076765909791, + "learning_rate": 1.8189105812005714e-05, + "loss": 1.5843726941966452e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00002, + "step": 418, + "tokens/total": 12647680, + "tokens/train_per_sec_per_gpu": 30.93, + "tokens/trainable": 191126 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.0011538882972672582, + "learning_rate": 1.802290048317732e-05, + "loss": 1.4711402400280349e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 419, + "tokens/total": 12677952, + "tokens/train_per_sec_per_gpu": 30.43, + "tokens/trainable": 191540 + }, + { + "epoch": 1.640625, + "grad_norm": 0.004467409569770098, + "learning_rate": 1.785823392239424e-05, + "loss": 4.823958806809969e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00005, + "step": 420, + "tokens/total": 12708416, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 192014 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.0012163098435848951, + "learning_rate": 1.7695112982104225e-05, + "loss": 1.1310569789202418e-05, + "memory/device_reserved (GiB)": 35.8, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00001, + "step": 421, + "tokens/total": 12738640, + "tokens/train_per_sec_per_gpu": 31.42, + "tokens/trainable": 192463 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.05586014315485954, + "learning_rate": 1.7533544450435433e-05, + "loss": 0.0002712146961130202, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00027, + "step": 422, + "tokens/total": 12769232, + "tokens/train_per_sec_per_gpu": 32.82, + "tokens/trainable": 192918 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.06670165061950684, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.0004127066058572382, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00041, + "step": 423, + "tokens/total": 12799648, + "tokens/train_per_sec_per_gpu": 37.51, + "tokens/trainable": 193425 + }, + { + "epoch": 1.65625, + "grad_norm": 0.019891072064638138, + "learning_rate": 1.721509144218405e-05, + "loss": 0.00023232161765918136, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00023, + "step": 424, + "tokens/total": 12829920, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 193898 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.006559237837791443, + "learning_rate": 1.705822021773101e-05, + "loss": 6.114803545642644e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 425, + "tokens/total": 12860112, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 194315 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.005621429067105055, + "learning_rate": 1.69029279056068e-05, + "loss": 6.088989175623283e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00006, + "step": 426, + "tokens/total": 12890320, + "tokens/train_per_sec_per_gpu": 35.06, + "tokens/trainable": 194796 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.0004855124861933291, + "learning_rate": 1.6749220968158415e-05, + "loss": 1.0911244316957891e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00001, + "step": 427, + "tokens/total": 12920656, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 195262 + }, + { + "epoch": 1.671875, + "grad_norm": 0.0006959444144740701, + "learning_rate": 1.659710580175893e-05, + "loss": 1.4198772987583652e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 428, + "tokens/total": 12951040, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 195680 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.00048281854833476245, + "learning_rate": 1.644658873654133e-05, + "loss": 1.0396102879894897e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00001, + "step": 429, + "tokens/total": 12981232, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 196145 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.007472805213183165, + "learning_rate": 1.629767603613508e-05, + "loss": 5.188406430534087e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00005, + "step": 430, + "tokens/total": 13011616, + "tokens/train_per_sec_per_gpu": 32.5, + "tokens/trainable": 196626 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.008452537469565868, + "learning_rate": 1.615037389740547e-05, + "loss": 2.078613033518195e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00002, + "step": 431, + "tokens/total": 13042208, + "tokens/train_per_sec_per_gpu": 31.22, + "tokens/trainable": 197067 + }, + { + "epoch": 1.6875, + "grad_norm": 0.0006947954534552991, + "learning_rate": 1.600468845019576e-05, + "loss": 1.0801179087138735e-05, + "memory/device_reserved (GiB)": 36.35, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00001, + "step": 432, + "tokens/total": 13072800, + "tokens/train_per_sec_per_gpu": 38.63, + "tokens/trainable": 197565 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.0020175843965262175, + "learning_rate": 1.5860625757072092e-05, + "loss": 1.564873855386395e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00002, + "step": 433, + "tokens/total": 13103328, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 198015 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.001340522081591189, + "learning_rate": 1.571819181307116e-05, + "loss": 1.7318390746368095e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 434, + "tokens/total": 13133472, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 198442 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.08985895663499832, + "learning_rate": 1.557739254545075e-05, + "loss": 0.0004795463755726814, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00048, + "step": 435, + "tokens/total": 13164064, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 198914 + }, + { + "epoch": 1.703125, + "grad_norm": 0.0005816713673993945, + "learning_rate": 1.543823381344311e-05, + "loss": 1.203996089316206e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00001, + "step": 436, + "tokens/total": 13194464, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 199355 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.2994021773338318, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.0055721537210047245, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00559, + "step": 437, + "tokens/total": 13224992, + "tokens/train_per_sec_per_gpu": 32.98, + "tokens/trainable": 199830 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.0006708212895318866, + "learning_rate": 1.5164861051607254e-05, + "loss": 9.834290722210426e-06, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00001, + "step": 438, + "tokens/total": 13255296, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 200338 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.004102411679923534, + "learning_rate": 1.5030658397935521e-05, + "loss": 2.766325997072272e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00003, + "step": 439, + "tokens/total": 13285344, + "tokens/train_per_sec_per_gpu": 32.91, + "tokens/trainable": 200785 + }, + { + "epoch": 1.71875, + "grad_norm": 0.0009204484522342682, + "learning_rate": 1.4898119031716104e-05, + "loss": 1.4549179468303919e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00001, + "step": 440, + "tokens/total": 13315632, + "tokens/train_per_sec_per_gpu": 36.38, + "tokens/trainable": 201267 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.0014299805043265224, + "learning_rate": 1.476724846845306e-05, + "loss": 2.365603722864762e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00002, + "step": 441, + "tokens/total": 13346048, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 201734 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.01532546803355217, + "learning_rate": 1.463805215420471e-05, + "loss": 9.99852636596188e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0001, + "step": 442, + "tokens/total": 13376304, + "tokens/train_per_sec_per_gpu": 30.36, + "tokens/trainable": 202174 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.18507054448127747, + "learning_rate": 1.451053546535705e-05, + "loss": 0.000987955485470593, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00099, + "step": 443, + "tokens/total": 13406720, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 202612 + }, + { + "epoch": 1.734375, + "grad_norm": 0.028212063014507294, + "learning_rate": 1.438470370840001e-05, + "loss": 0.00025927621754817665, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00026, + "step": 444, + "tokens/total": 13436464, + "tokens/train_per_sec_per_gpu": 30.96, + "tokens/trainable": 203020 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.0004877297324128449, + "learning_rate": 1.4260562119706606e-05, + "loss": 1.2777243682648987e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00001, + "step": 445, + "tokens/total": 13466672, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 203479 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.02534927800297737, + "learning_rate": 1.413811586531508e-05, + "loss": 3.3006788726197556e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 446, + "tokens/total": 13496736, + "tokens/train_per_sec_per_gpu": 35.38, + "tokens/trainable": 203946 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.0006437217816710472, + "learning_rate": 1.4017370040713884e-05, + "loss": 1.1569853086257353e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00001, + "step": 447, + "tokens/total": 13526928, + "tokens/train_per_sec_per_gpu": 37.38, + "tokens/trainable": 204382 + }, + { + "epoch": 1.75, + "grad_norm": 0.0014389768475666642, + "learning_rate": 1.3898329670629645e-05, + "loss": 2.1063802705612034e-05, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00002, + "step": 448, + "tokens/total": 13557392, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 204830 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.016985442489385605, + "learning_rate": 1.3780999708818058e-05, + "loss": 0.00012392934877425432, + "memory/device_reserved (GiB)": 36.8, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00012, + "step": 449, + "tokens/total": 13587776, + "tokens/train_per_sec_per_gpu": 33.74, + "tokens/trainable": 205300 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.008965734392404556, + "learning_rate": 1.3665385037857758e-05, + "loss": 6.039683285052888e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 450, + "tokens/total": 13618112, + "tokens/train_per_sec_per_gpu": 30.72, + "tokens/trainable": 205726 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.5729678273200989, + "learning_rate": 1.3551490468947126e-05, + "loss": 0.00580303929746151, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00582, + "step": 451, + "tokens/total": 13648512, + "tokens/train_per_sec_per_gpu": 31.74, + "tokens/trainable": 206163 + }, + { + "epoch": 1.765625, + "grad_norm": 0.0277901329100132, + "learning_rate": 1.3439320741704075e-05, + "loss": 9.134741412708536e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 452, + "tokens/total": 13678704, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 206619 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.3788454532623291, + "learning_rate": 1.3328880523968808e-05, + "loss": 0.002095964504405856, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.0021, + "step": 453, + "tokens/total": 13709232, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 207101 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.00038695961120538414, + "learning_rate": 1.3220174411609587e-05, + "loss": 9.780245818546973e-06, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00001, + "step": 454, + "tokens/total": 13739600, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 207546 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.01514297816902399, + "learning_rate": 1.3113206928331471e-05, + "loss": 7.800674939062446e-05, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00008, + "step": 455, + "tokens/total": 13769936, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 207989 + }, + { + "epoch": 1.78125, + "grad_norm": 0.0018013569060713053, + "learning_rate": 1.300798252548806e-05, + "loss": 2.1191644918872043e-05, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00002, + "step": 456, + "tokens/total": 13800240, + "tokens/train_per_sec_per_gpu": 34.06, + "tokens/trainable": 208402 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.0006610782584175467, + "learning_rate": 1.2904505581896265e-05, + "loss": 1.3429066711978521e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00001, + "step": 457, + "tokens/total": 13830512, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 208904 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.0005749509437009692, + "learning_rate": 1.2802780403654082e-05, + "loss": 1.3295277312863618e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00001, + "step": 458, + "tokens/total": 13860832, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 209382 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.00635139923542738, + "learning_rate": 1.2702811223961408e-05, + "loss": 5.09147321281489e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00005, + "step": 459, + "tokens/total": 13890992, + "tokens/train_per_sec_per_gpu": 34.01, + "tokens/trainable": 209833 + }, + { + "epoch": 1.796875, + "grad_norm": 0.013369702734053135, + "learning_rate": 1.2604602202943861e-05, + "loss": 0.00011205296323169023, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00011, + "step": 460, + "tokens/total": 13921424, + "tokens/train_per_sec_per_gpu": 32.79, + "tokens/trainable": 210285 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.016762293875217438, + "learning_rate": 1.2508157427479686e-05, + "loss": 0.00019885516667272896, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.0002, + "step": 461, + "tokens/total": 13951504, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 210770 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.006627500057220459, + "learning_rate": 1.2413480911029655e-05, + "loss": 2.5290853955084458e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00003, + "step": 462, + "tokens/total": 13979728, + "tokens/train_per_sec_per_gpu": 30.5, + "tokens/trainable": 211194 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.003884925739839673, + "learning_rate": 1.2320576593470082e-05, + "loss": 2.766420948319137e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00003, + "step": 463, + "tokens/total": 14009936, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 211656 + }, + { + "epoch": 1.8125, + "grad_norm": 0.000377490563550964, + "learning_rate": 1.2229448340928828e-05, + "loss": 9.118146408582106e-06, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00001, + "step": 464, + "tokens/total": 14039872, + "tokens/train_per_sec_per_gpu": 29.49, + "tokens/trainable": 212086 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.01634475402534008, + "learning_rate": 1.2140099945624458e-05, + "loss": 7.731329969828948e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00008, + "step": 465, + "tokens/total": 14070096, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 212571 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.5773619413375854, + "learning_rate": 1.205253512570841e-05, + "loss": 0.008904650807380676, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00894, + "step": 466, + "tokens/total": 14100192, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 213010 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.005223860964179039, + "learning_rate": 1.1966757525110255e-05, + "loss": 3.063467738684267e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00003, + "step": 467, + "tokens/total": 14130448, + "tokens/train_per_sec_per_gpu": 31.72, + "tokens/trainable": 213449 + }, + { + "epoch": 1.828125, + "grad_norm": 0.00045610740198753774, + "learning_rate": 1.1882770713386095e-05, + "loss": 1.0150557500310242e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00001, + "step": 468, + "tokens/total": 14160480, + "tokens/train_per_sec_per_gpu": 35.84, + "tokens/trainable": 213897 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.005935850087553263, + "learning_rate": 1.180057818556998e-05, + "loss": 4.396100121084601e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00004, + "step": 469, + "tokens/total": 14191152, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 214365 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.0010427028173580766, + "learning_rate": 1.1720183362028494e-05, + "loss": 2.14513493119739e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00002, + "step": 470, + "tokens/total": 14219392, + "tokens/train_per_sec_per_gpu": 35.86, + "tokens/trainable": 214799 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.006076959893107414, + "learning_rate": 1.1641589588318387e-05, + "loss": 1.3421818948700093e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00001, + "step": 471, + "tokens/total": 14249920, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 215291 + }, + { + "epoch": 1.84375, + "grad_norm": 0.0019434966379776597, + "learning_rate": 1.1564800135047418e-05, + "loss": 3.379978079465218e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00003, + "step": 472, + "tokens/total": 14280048, + "tokens/train_per_sec_per_gpu": 34.1, + "tokens/trainable": 215740 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.01917141303420067, + "learning_rate": 1.148981819773816e-05, + "loss": 8.30673161544837e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 473, + "tokens/total": 14310432, + "tokens/train_per_sec_per_gpu": 34.88, + "tokens/trainable": 216193 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.027403976768255234, + "learning_rate": 1.1416646896695086e-05, + "loss": 5.644366319756955e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00006, + "step": 474, + "tokens/total": 14340496, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 216633 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.0030669430270791054, + "learning_rate": 1.1345289276874717e-05, + "loss": 3.4152639273088425e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00003, + "step": 475, + "tokens/total": 14370704, + "tokens/train_per_sec_per_gpu": 30.62, + "tokens/trainable": 217039 + }, + { + "epoch": 1.859375, + "grad_norm": 0.0022093369625508785, + "learning_rate": 1.1275748307758873e-05, + "loss": 3.003427991643548e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 476, + "tokens/total": 14401120, + "tokens/train_per_sec_per_gpu": 37.85, + "tokens/trainable": 217529 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.007151913829147816, + "learning_rate": 1.1208026883231147e-05, + "loss": 4.787252328242175e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00005, + "step": 477, + "tokens/total": 14431344, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 217987 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.005195859353989363, + "learning_rate": 1.1142127821456433e-05, + "loss": 6.279069930315018e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00006, + "step": 478, + "tokens/total": 14461952, + "tokens/train_per_sec_per_gpu": 32.73, + "tokens/trainable": 218460 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.0024305693805217743, + "learning_rate": 1.1078053864763674e-05, + "loss": 4.7390385589096695e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00005, + "step": 479, + "tokens/total": 14491968, + "tokens/train_per_sec_per_gpu": 32.87, + "tokens/trainable": 218908 + }, + { + "epoch": 1.875, + "grad_norm": 0.0007113535539247096, + "learning_rate": 1.1015807679531756e-05, + "loss": 1.3227703675511293e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00001, + "step": 480, + "tokens/total": 14522528, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 219349 + }, + { + "epoch": 1.87890625, + "grad_norm": 0.0017267197836190462, + "learning_rate": 1.0955391856078528e-05, + "loss": 1.1387310223653913e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00001, + "step": 481, + "tokens/total": 14552544, + "tokens/train_per_sec_per_gpu": 29.38, + "tokens/trainable": 219775 + }, + { + "epoch": 1.8828125, + "grad_norm": 0.01939094066619873, + "learning_rate": 1.0896808908553007e-05, + "loss": 7.642045238753781e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 482, + "tokens/total": 14583040, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 220283 + }, + { + "epoch": 1.88671875, + "grad_norm": 0.004050788469612598, + "learning_rate": 1.0840061274830763e-05, + "loss": 5.199073348194361e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00005, + "step": 483, + "tokens/total": 14613136, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 220713 + }, + { + "epoch": 1.890625, + "grad_norm": 0.0024211062118411064, + "learning_rate": 1.0785151316412473e-05, + "loss": 2.763515840342734e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00003, + "step": 484, + "tokens/total": 14643312, + "tokens/train_per_sec_per_gpu": 35.18, + "tokens/trainable": 221155 + }, + { + "epoch": 1.89453125, + "grad_norm": 0.015418405644595623, + "learning_rate": 1.0732081318325639e-05, + "loss": 7.891400309745222e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00008, + "step": 485, + "tokens/total": 14673472, + "tokens/train_per_sec_per_gpu": 31.78, + "tokens/trainable": 221604 + }, + { + "epoch": 1.8984375, + "grad_norm": 0.06489475816488266, + "learning_rate": 1.0680853489029501e-05, + "loss": 0.00017086212756112218, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00017, + "step": 486, + "tokens/total": 14703984, + "tokens/train_per_sec_per_gpu": 30.42, + "tokens/trainable": 222040 + }, + { + "epoch": 1.90234375, + "grad_norm": 0.000592821161262691, + "learning_rate": 1.0631469960323152e-05, + "loss": 1.0421663318993524e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00001, + "step": 487, + "tokens/total": 14733968, + "tokens/train_per_sec_per_gpu": 32.16, + "tokens/trainable": 222472 + }, + { + "epoch": 1.90625, + "grad_norm": 0.00045594078255817294, + "learning_rate": 1.0583932787256783e-05, + "loss": 1.1962510143348482e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00001, + "step": 488, + "tokens/total": 14764384, + "tokens/train_per_sec_per_gpu": 36.41, + "tokens/trainable": 222965 + }, + { + "epoch": 1.91015625, + "grad_norm": 0.001553925801999867, + "learning_rate": 1.0538243948046206e-05, + "loss": 2.3432628950104117e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00002, + "step": 489, + "tokens/total": 14794656, + "tokens/train_per_sec_per_gpu": 31.93, + "tokens/trainable": 223418 + }, + { + "epoch": 1.9140625, + "grad_norm": 0.0029810869600623846, + "learning_rate": 1.0494405343990523e-05, + "loss": 2.7821279218187556e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00003, + "step": 490, + "tokens/total": 14824768, + "tokens/train_per_sec_per_gpu": 29.52, + "tokens/trainable": 223862 + }, + { + "epoch": 1.91796875, + "grad_norm": 0.00159536674618721, + "learning_rate": 1.0452418799392985e-05, + "loss": 2.1142092009540647e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00002, + "step": 491, + "tokens/total": 14855040, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 224319 + }, + { + "epoch": 1.921875, + "grad_norm": 0.003593488596379757, + "learning_rate": 1.0412286061485102e-05, + "loss": 3.937885048799217e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00004, + "step": 492, + "tokens/total": 14885376, + "tokens/train_per_sec_per_gpu": 36.5, + "tokens/trainable": 224800 + }, + { + "epoch": 1.92578125, + "grad_norm": 0.0018238810589537024, + "learning_rate": 1.03740088003539e-05, + "loss": 2.2674561478197575e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 493, + "tokens/total": 14915744, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 225282 + }, + { + "epoch": 1.9296875, + "grad_norm": 0.0027770590968430042, + "learning_rate": 1.0337588608872463e-05, + "loss": 1.480587525293231e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00001, + "step": 494, + "tokens/total": 14946176, + "tokens/train_per_sec_per_gpu": 32.91, + "tokens/trainable": 225731 + }, + { + "epoch": 1.93359375, + "grad_norm": 0.03237319365143776, + "learning_rate": 1.0303027002633622e-05, + "loss": 0.00021609703253488988, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00022, + "step": 495, + "tokens/total": 14976256, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 226170 + }, + { + "epoch": 1.9375, + "grad_norm": 0.00870265532284975, + "learning_rate": 1.0270325419886884e-05, + "loss": 5.133711965754628e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00005, + "step": 496, + "tokens/total": 15006528, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 226634 + }, + { + "epoch": 1.94140625, + "grad_norm": 0.0011709899408742785, + "learning_rate": 1.0239485221478599e-05, + "loss": 2.115583083650563e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00002, + "step": 497, + "tokens/total": 15037024, + "tokens/train_per_sec_per_gpu": 33.07, + "tokens/trainable": 227065 + }, + { + "epoch": 1.9453125, + "grad_norm": 0.0010721588041633368, + "learning_rate": 1.0210507690795292e-05, + "loss": 1.7287293303525075e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00002, + "step": 498, + "tokens/total": 15067328, + "tokens/train_per_sec_per_gpu": 37.81, + "tokens/trainable": 227530 + }, + { + "epoch": 1.94921875, + "grad_norm": 0.06478072702884674, + "learning_rate": 1.0183394033710305e-05, + "loss": 0.000421134231146425, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00042, + "step": 499, + "tokens/total": 15097632, + "tokens/train_per_sec_per_gpu": 36.76, + "tokens/trainable": 228005 + }, + { + "epoch": 1.953125, + "grad_norm": 0.01197084691375494, + "learning_rate": 1.0158145378533583e-05, + "loss": 0.00013856604346074164, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00014, + "step": 500, + "tokens/total": 15127984, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 228455 + }, + { + "epoch": 1.95703125, + "grad_norm": 0.0047052945010364056, + "learning_rate": 1.0134762775964726e-05, + "loss": 3.4345306630712e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00003, + "step": 501, + "tokens/total": 15158336, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 228926 + }, + { + "epoch": 1.9609375, + "grad_norm": 0.01203920878469944, + "learning_rate": 1.0113247199049278e-05, + "loss": 5.282826168695465e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.00005, + "step": 502, + "tokens/total": 15189120, + "tokens/train_per_sec_per_gpu": 32.25, + "tokens/trainable": 229366 + }, + { + "epoch": 1.96484375, + "grad_norm": 0.06692500412464142, + "learning_rate": 1.0093599543138205e-05, + "loss": 3.876092523569241e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00004, + "step": 503, + "tokens/total": 15219600, + "tokens/train_per_sec_per_gpu": 32.93, + "tokens/trainable": 229813 + }, + { + "epoch": 1.96875, + "grad_norm": 0.05963515117764473, + "learning_rate": 1.0075820625850675e-05, + "loss": 0.0003227375273127109, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00032, + "step": 504, + "tokens/total": 15250016, + "tokens/train_per_sec_per_gpu": 39.03, + "tokens/trainable": 230317 + }, + { + "epoch": 1.97265625, + "grad_norm": 0.007018797565251589, + "learning_rate": 1.0059911187040013e-05, + "loss": 5.134383536642417e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00005, + "step": 505, + "tokens/total": 15280576, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 230805 + }, + { + "epoch": 1.9765625, + "grad_norm": 0.006387659814208746, + "learning_rate": 1.0045871888762893e-05, + "loss": 4.712470399681479e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00005, + "step": 506, + "tokens/total": 15310800, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 231255 + }, + { + "epoch": 1.98046875, + "grad_norm": 0.001986650750041008, + "learning_rate": 1.003370331525184e-05, + "loss": 2.7338617655914277e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00003, + "step": 507, + "tokens/total": 15341408, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 231744 + }, + { + "epoch": 1.984375, + "grad_norm": 0.0114446384832263, + "learning_rate": 1.002340597289085e-05, + "loss": 3.070381353609264e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.33, + "memory/max_allocated (GiB)": 33.33, + "ppl": 1.00003, + "step": 508, + "tokens/total": 15369648, + "tokens/train_per_sec_per_gpu": 33.13, + "tokens/trainable": 232170 + }, + { + "epoch": 1.98828125, + "grad_norm": 0.0035379640758037567, + "learning_rate": 1.0014980290194387e-05, + "loss": 2.992351073771715e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00003, + "step": 509, + "tokens/total": 15400480, + "tokens/train_per_sec_per_gpu": 37.24, + "tokens/trainable": 232678 + }, + { + "epoch": 1.9921875, + "grad_norm": 0.13356627523899078, + "learning_rate": 1.0008426617789489e-05, + "loss": 0.0014438352081924677, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00144, + "step": 510, + "tokens/total": 15430848, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 233167 + }, + { + "epoch": 1.99609375, + "grad_norm": 0.0027949470095336437, + "learning_rate": 1.0003745228401215e-05, + "loss": 1.4765095329494216e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00001, + "step": 511, + "tokens/total": 15461216, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 233607 + }, + { + "epoch": 2.0, + "grad_norm": 0.328495055437088, + "learning_rate": 1.0000936316841296e-05, + "loss": 0.0018887810874730349, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00189, + "step": 512, + "tokens/total": 15491744, + "tokens/train_per_sec_per_gpu": 36.34, + "tokens/trainable": 234076 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 1.0515151950660495e+18, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-512/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..b69bdd579e8b905662f522a6ee5c224db166d819 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b584e3cf8f378cc0d06c9d8e81b75d4ebec54e9ef2e3bc83848da1c57396f4f +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..0a677dc83aa328b88094ccbaad2789d815cad0e6 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7f8513c58f406f91f4e3f9606a17ac066ff4724ad1843f31a0343e812b8d82fa +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..29fe8c0957f6a29bef030536a0d836c14dd13199 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21d00dc563d713980dd7a1d4b126d93ab060fe78b7ba0c4c92f92cd25a992569 +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..34b7fc400f1006f136b1c89d6c58cf3d17830d72 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c053d5c6e293421f32ec74be7fd526b45c9c8622 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/tokens_state.json @@ -0,0 +1 @@ +{"total": 1940848, "trainable": 29183} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..ccac86a824e43e1f219b7b1d1e4785c142cbb4b8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/trainer_state.json @@ -0,0 +1,930 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.25, + "eval_steps": 500, + "global_step": 64, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.3173669557885491e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-64/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/README.md b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/adapter_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..90f169409a88a2132054b4d8557cde5c40b8ffa4 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "down_proj", + "v_proj", + "up_proj", + "k_proj", + "gate_proj", + "q_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/adapter_model.safetensors b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..1c217403883e1017df991dcf6cab7cfcf405e1ee --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c9a211aba75f9a340dc57de855c8e2b458e1b586134e0dc1ad3988647f0aa4e9 +size 547777976 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/chat_template.jinja b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/optimizer.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..353dafac2b68040c4882e1ff868db9b6913a1bfc --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c5a9fb4a808c243ca1d455c97cb546577f69e79453fc34bf8fae30def5fa9a8 +size 1048106435 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/rng_state.pth b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..c033f4347b764091af577cf5043dff72a382d3f5 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:58b2d07d478cf0253d5619ad2e9d23f5e40a65de8495503676e2505cd9d2856a +size 14645 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/scheduler.pt b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..433f82b14836de77d366e2c961bc5bdacdc9eb63 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0 +size 1465 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer_config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokens_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..b4a8c8f671c130f9a8c18d776fdd5f367198489d --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/tokens_state.json @@ -0,0 +1 @@ +{"total": 2906288, "trainable": 43824} \ No newline at end of file diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/trainer_state.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..a68151c2925f4d9fc2463142c8251609bf1177d0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/trainer_state.json @@ -0,0 +1,1378 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.375, + "eval_steps": 500, + "global_step": 96, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 1.000339388847351, + "learning_rate": 0.0, + "loss": 0.11221420764923096, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.11875, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 27.52, + "tokens/trainable": 417 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8961698412895203, + "learning_rate": 4.000000000000001e-06, + "loss": 0.11656901985406876, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.12364, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 873 + }, + { + "epoch": 0.01171875, + "grad_norm": 2.5287365913391113, + "learning_rate": 8.000000000000001e-06, + "loss": 0.11233991384506226, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.11889, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 1353 + }, + { + "epoch": 0.015625, + "grad_norm": 1.3438372611999512, + "learning_rate": 1.2e-05, + "loss": 0.10132055729627609, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.10663, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.26, + "tokens/trainable": 1778 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.0121135711669922, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.11346302926540375, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.12015, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 36.72, + "tokens/trainable": 2261 + }, + { + "epoch": 0.0234375, + "grad_norm": 3.9203455448150635, + "learning_rate": 2e-05, + "loss": 0.09780596196651459, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.10275, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 32.9, + "tokens/trainable": 2705 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.63088059425354, + "learning_rate": 2.4e-05, + "loss": 0.06366116553544998, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.06573, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 33.89, + "tokens/trainable": 3136 + }, + { + "epoch": 0.03125, + "grad_norm": 1.9995646476745605, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.0759042352437973, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07886, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.22, + "tokens/trainable": 3579 + }, + { + "epoch": 0.03515625, + "grad_norm": 1.0774554014205933, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04145222157239914, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04232, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.14, + "tokens/trainable": 4000 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.9648898839950562, + "learning_rate": 3.6e-05, + "loss": 0.032875221222639084, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03342, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 37.67, + "tokens/trainable": 4474 + }, + { + "epoch": 0.04296875, + "grad_norm": 0.5937227606773376, + "learning_rate": 4e-05, + "loss": 0.023198626935482025, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.02347, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 4942 + }, + { + "epoch": 0.046875, + "grad_norm": 5.294079303741455, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.0510174036026001, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.05234, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 5424 + }, + { + "epoch": 0.05078125, + "grad_norm": 2.7166640758514404, + "learning_rate": 4.8e-05, + "loss": 0.06956956535577774, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.07205, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.23, + "tokens/trainable": 5855 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.1496169567108154, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.06250961124897003, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0645, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 6344 + }, + { + "epoch": 0.05859375, + "grad_norm": 3.515660047531128, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.09882807731628418, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.10388, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 6825 + }, + { + "epoch": 0.0625, + "grad_norm": 2.3756372928619385, + "learning_rate": 6e-05, + "loss": 0.0719379335641861, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.07459, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 30.01, + "tokens/trainable": 7230 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.7220206260681152, + "learning_rate": 6.400000000000001e-05, + "loss": 0.016590356826782227, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01673, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 7699 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.5261812210083008, + "learning_rate": 6.800000000000001e-05, + "loss": 0.030052796006202698, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.03051, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 33.85, + "tokens/trainable": 8184 + }, + { + "epoch": 0.07421875, + "grad_norm": 0.6416668891906738, + "learning_rate": 7.2e-05, + "loss": 0.013492297381162643, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01358, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 34.91, + "tokens/trainable": 8675 + }, + { + "epoch": 0.078125, + "grad_norm": 1.2296113967895508, + "learning_rate": 7.6e-05, + "loss": 0.038845278322696686, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.03961, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 39.72, + "tokens/trainable": 9189 + }, + { + "epoch": 0.08203125, + "grad_norm": 0.8426793813705444, + "learning_rate": 8e-05, + "loss": 0.01696154847741127, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01711, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 9614 + }, + { + "epoch": 0.0859375, + "grad_norm": 10.310348510742188, + "learning_rate": 8.4e-05, + "loss": 0.06855659186840057, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.07096, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 10086 + }, + { + "epoch": 0.08984375, + "grad_norm": 1.412604808807373, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04118483513593674, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04204, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 10535 + }, + { + "epoch": 0.09375, + "grad_norm": 1.4078032970428467, + "learning_rate": 9.200000000000001e-05, + "loss": 0.025748739019036293, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02608, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 31.6, + "tokens/trainable": 10993 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0874605178833008, + "learning_rate": 9.6e-05, + "loss": 0.03088424727320671, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.03137, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.03, + "tokens/trainable": 11417 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.2574082612991333, + "learning_rate": 0.0001, + "loss": 0.03432866185903549, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.03492, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 11874 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.30990996956825256, + "learning_rate": 9.99990636831587e-05, + "loss": 0.010750269517302513, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01081, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 31.64, + "tokens/trainable": 12316 + }, + { + "epoch": 0.109375, + "grad_norm": 0.5124737620353699, + "learning_rate": 9.999625477159879e-05, + "loss": 0.01967313140630722, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01987, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 37.86, + "tokens/trainable": 12800 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.2975701093673706, + "learning_rate": 9.999157338221051e-05, + "loss": 0.0434565395116806, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04441, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 35.48, + "tokens/trainable": 13270 + }, + { + "epoch": 0.1171875, + "grad_norm": 0.4117317795753479, + "learning_rate": 9.998501970980562e-05, + "loss": 0.013137167319655418, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01322, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.39, + "tokens/trainable": 13752 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.5875954627990723, + "learning_rate": 9.997659402710915e-05, + "loss": 0.01785588450729847, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01802, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 14218 + }, + { + "epoch": 0.125, + "grad_norm": 0.7800514698028564, + "learning_rate": 9.996629668474818e-05, + "loss": 0.011262159794569016, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01133, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 14655 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.0509369373321533, + "learning_rate": 9.995412811123711e-05, + "loss": 0.04942459613084793, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05067, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 15087 + }, + { + "epoch": 0.1328125, + "grad_norm": 0.6234491467475891, + "learning_rate": 9.994008881295999e-05, + "loss": 0.028202136978507042, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.0286, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 15548 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.6843816041946411, + "learning_rate": 9.992417937414932e-05, + "loss": 0.02795713022351265, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02835, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 33.77, + "tokens/trainable": 16037 + }, + { + "epoch": 0.140625, + "grad_norm": 0.29489848017692566, + "learning_rate": 9.99064004568618e-05, + "loss": 0.010517537593841553, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01057, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.14, + "tokens/trainable": 16475 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7335977554321289, + "learning_rate": 9.988675280095074e-05, + "loss": 0.018923252820968628, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0191, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 16919 + }, + { + "epoch": 0.1484375, + "grad_norm": 0.5587971210479736, + "learning_rate": 9.986523722403528e-05, + "loss": 0.020957889035344124, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02118, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 28.97, + "tokens/trainable": 17333 + }, + { + "epoch": 0.15234375, + "grad_norm": 1.7112773656845093, + "learning_rate": 9.984185462146642e-05, + "loss": 0.045844268053770065, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.04691, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.19, + "tokens/trainable": 17806 + }, + { + "epoch": 0.15625, + "grad_norm": 1.2373344898223877, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04578991234302521, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04685, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 18290 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.8950033187866211, + "learning_rate": 9.978949230920472e-05, + "loss": 0.02866373583674431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02908, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 18747 + }, + { + "epoch": 0.1640625, + "grad_norm": 1.9967094659805298, + "learning_rate": 9.976051477852141e-05, + "loss": 0.02229372225701809, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02254, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 31.91, + "tokens/trainable": 19172 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.1990034282207489, + "learning_rate": 9.972967458011312e-05, + "loss": 0.0059759169816970825, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.00599, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.32, + "tokens/trainable": 19622 + }, + { + "epoch": 0.171875, + "grad_norm": 0.47151026129722595, + "learning_rate": 9.96969729973664e-05, + "loss": 0.011378830298781395, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01144, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 20073 + }, + { + "epoch": 0.17578125, + "grad_norm": 0.5381685495376587, + "learning_rate": 9.966241139112754e-05, + "loss": 0.02141587622463703, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02165, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.57, + "tokens/trainable": 20497 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.0174895524978638, + "learning_rate": 9.96259911996461e-05, + "loss": 0.02236388623714447, + "memory/device_reserved (GiB)": 36.03, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02262, + "step": 46, + "tokens/total": 1394368, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 20940 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5081813931465149, + "learning_rate": 9.958771393851491e-05, + "loss": 0.010915009304881096, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.01097, + "step": 47, + "tokens/total": 1425024, + "tokens/train_per_sec_per_gpu": 35.04, + "tokens/trainable": 21405 + }, + { + "epoch": 0.1875, + "grad_norm": 1.4684470891952515, + "learning_rate": 9.954758120060702e-05, + "loss": 0.009690589271485806, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00974, + "step": 48, + "tokens/total": 1455392, + "tokens/train_per_sec_per_gpu": 35.13, + "tokens/trainable": 21828 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.49191179871559143, + "learning_rate": 9.950559465600948e-05, + "loss": 0.010902130044996738, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01096, + "step": 49, + "tokens/total": 1485616, + "tokens/train_per_sec_per_gpu": 29.55, + "tokens/trainable": 22241 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.44692206382751465, + "learning_rate": 9.946175605195379e-05, + "loss": 0.006928037852048874, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00695, + "step": 50, + "tokens/total": 1515984, + "tokens/train_per_sec_per_gpu": 36.48, + "tokens/trainable": 22688 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.37972721457481384, + "learning_rate": 9.941606721274322e-05, + "loss": 0.00806603953242302, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0081, + "step": 51, + "tokens/total": 1546224, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 23136 + }, + { + "epoch": 0.203125, + "grad_norm": 1.0114665031433105, + "learning_rate": 9.936853003967685e-05, + "loss": 0.013657575473189354, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01375, + "step": 52, + "tokens/total": 1576528, + "tokens/train_per_sec_per_gpu": 36.95, + "tokens/trainable": 23631 + }, + { + "epoch": 0.20703125, + "grad_norm": 0.6438844799995422, + "learning_rate": 9.93191465109705e-05, + "loss": 0.05178550258278847, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.05315, + "step": 53, + "tokens/total": 1606768, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 24096 + }, + { + "epoch": 0.2109375, + "grad_norm": 3.0008814334869385, + "learning_rate": 9.926791868167438e-05, + "loss": 0.010915370658040047, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.01098, + "step": 54, + "tokens/total": 1636640, + "tokens/train_per_sec_per_gpu": 37.95, + "tokens/trainable": 24569 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.9377115368843079, + "learning_rate": 9.921484868358753e-05, + "loss": 0.019770421087741852, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01997, + "step": 55, + "tokens/total": 1667136, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 25067 + }, + { + "epoch": 0.21875, + "grad_norm": 0.23384763300418854, + "learning_rate": 9.915993872516924e-05, + "loss": 0.004394679330289364, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0044, + "step": 56, + "tokens/total": 1697472, + "tokens/train_per_sec_per_gpu": 36.68, + "tokens/trainable": 25541 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.7558876276016235, + "learning_rate": 9.9103191091447e-05, + "loss": 0.02614775486290455, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.02649, + "step": 57, + "tokens/total": 1727680, + "tokens/train_per_sec_per_gpu": 33.29, + "tokens/trainable": 25965 + }, + { + "epoch": 0.2265625, + "grad_norm": 0.4615328311920166, + "learning_rate": 9.904460814392147e-05, + "loss": 0.01031234860420227, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01037, + "step": 58, + "tokens/total": 1758192, + "tokens/train_per_sec_per_gpu": 34.0, + "tokens/trainable": 26409 + }, + { + "epoch": 0.23046875, + "grad_norm": 0.40572404861450195, + "learning_rate": 9.898419232046825e-05, + "loss": 0.011052684858441353, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01111, + "step": 59, + "tokens/total": 1788816, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 26883 + }, + { + "epoch": 0.234375, + "grad_norm": 0.7271711826324463, + "learning_rate": 9.892194613523633e-05, + "loss": 0.01451511587947607, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01462, + "step": 60, + "tokens/total": 1819056, + "tokens/train_per_sec_per_gpu": 33.87, + "tokens/trainable": 27323 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.14707089960575104, + "learning_rate": 9.885787217854357e-05, + "loss": 0.004126548767089844, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00414, + "step": 61, + "tokens/total": 1849664, + "tokens/train_per_sec_per_gpu": 39.34, + "tokens/trainable": 27830 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.4162050485610962, + "learning_rate": 9.879197311676887e-05, + "loss": 0.009982893243432045, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.01003, + "step": 62, + "tokens/total": 1880272, + "tokens/train_per_sec_per_gpu": 32.46, + "tokens/trainable": 28281 + }, + { + "epoch": 0.24609375, + "grad_norm": 0.563757598400116, + "learning_rate": 9.872425169224113e-05, + "loss": 0.01410748716443777, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01421, + "step": 63, + "tokens/total": 1910752, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 28742 + }, + { + "epoch": 0.25, + "grad_norm": 0.4908931255340576, + "learning_rate": 9.865471072312528e-05, + "loss": 0.011950287967920303, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01202, + "step": 64, + "tokens/total": 1940848, + "tokens/train_per_sec_per_gpu": 33.76, + "tokens/trainable": 29183 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.3164263963699341, + "learning_rate": 9.858335310330492e-05, + "loss": 0.006426030304282904, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00645, + "step": 65, + "tokens/total": 1971344, + "tokens/train_per_sec_per_gpu": 31.69, + "tokens/trainable": 29648 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.4743383824825287, + "learning_rate": 9.851018180226185e-05, + "loss": 0.00986128207296133, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00991, + "step": 66, + "tokens/total": 2001712, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 30075 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.12894771993160248, + "learning_rate": 9.843519986495259e-05, + "loss": 0.002054001437500119, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00206, + "step": 67, + "tokens/total": 2029936, + "tokens/train_per_sec_per_gpu": 37.69, + "tokens/trainable": 30546 + }, + { + "epoch": 0.265625, + "grad_norm": 0.5559160709381104, + "learning_rate": 9.835841041168162e-05, + "loss": 0.007931358180940151, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00796, + "step": 68, + "tokens/total": 2060384, + "tokens/train_per_sec_per_gpu": 35.27, + "tokens/trainable": 31025 + }, + { + "epoch": 0.26953125, + "grad_norm": 1.4860860109329224, + "learning_rate": 9.82798166379715e-05, + "loss": 0.029434870928525925, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02987, + "step": 69, + "tokens/total": 2090688, + "tokens/train_per_sec_per_gpu": 34.03, + "tokens/trainable": 31505 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.2197936624288559, + "learning_rate": 9.819942181443002e-05, + "loss": 0.0028239269740879536, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00283, + "step": 70, + "tokens/total": 2120848, + "tokens/train_per_sec_per_gpu": 32.39, + "tokens/trainable": 31956 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.20881125330924988, + "learning_rate": 9.811722928661392e-05, + "loss": 0.001874964451417327, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.00188, + "step": 71, + "tokens/total": 2149120, + "tokens/train_per_sec_per_gpu": 33.46, + "tokens/trainable": 32374 + }, + { + "epoch": 0.28125, + "grad_norm": 0.335940957069397, + "learning_rate": 9.803324247488975e-05, + "loss": 0.003750729141756892, + "memory/device_reserved (GiB)": 35.64, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00376, + "step": 72, + "tokens/total": 2179472, + "tokens/train_per_sec_per_gpu": 34.67, + "tokens/trainable": 32833 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.6529806852340698, + "learning_rate": 9.794746487429161e-05, + "loss": 0.010189465247094631, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01024, + "step": 73, + "tokens/total": 2209984, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 33304 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.7839381098747253, + "learning_rate": 9.785990005437554e-05, + "loss": 0.00870432797819376, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00874, + "step": 74, + "tokens/total": 2240512, + "tokens/train_per_sec_per_gpu": 32.54, + "tokens/trainable": 33771 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.3333408832550049, + "learning_rate": 9.777055165907117e-05, + "loss": 0.007429605815559626, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00746, + "step": 75, + "tokens/total": 2270608, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 34186 + }, + { + "epoch": 0.296875, + "grad_norm": 0.47243013978004456, + "learning_rate": 9.767942340652993e-05, + "loss": 0.009635023772716522, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00968, + "step": 76, + "tokens/total": 2300864, + "tokens/train_per_sec_per_gpu": 35.05, + "tokens/trainable": 34667 + }, + { + "epoch": 0.30078125, + "grad_norm": 1.4933921098709106, + "learning_rate": 9.758651908897035e-05, + "loss": 0.04087181016802788, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.04172, + "step": 77, + "tokens/total": 2330960, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 35078 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.6791361570358276, + "learning_rate": 9.749184257252033e-05, + "loss": 0.013059539720416069, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01315, + "step": 78, + "tokens/total": 2361584, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 35550 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.8410643935203552, + "learning_rate": 9.739539779705614e-05, + "loss": 0.02711997739970684, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02749, + "step": 79, + "tokens/total": 2391936, + "tokens/train_per_sec_per_gpu": 29.62, + "tokens/trainable": 35977 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3738754689693451, + "learning_rate": 9.729718877603861e-05, + "loss": 0.017121555283665657, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01727, + "step": 80, + "tokens/total": 2422432, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 36462 + }, + { + "epoch": 0.31640625, + "grad_norm": 1.0237257480621338, + "learning_rate": 9.719721959634592e-05, + "loss": 0.019660785794258118, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01986, + "step": 81, + "tokens/total": 2452752, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 36919 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3411601185798645, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011061297729611397, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01112, + "step": 82, + "tokens/total": 2481216, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 37379 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.41286832094192505, + "learning_rate": 9.699201747451195e-05, + "loss": 0.011656982824206352, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01173, + "step": 83, + "tokens/total": 2511488, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 37832 + }, + { + "epoch": 0.328125, + "grad_norm": 0.6960376501083374, + "learning_rate": 9.688679307166854e-05, + "loss": 0.024275368079543114, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02457, + "step": 84, + "tokens/total": 2541984, + "tokens/train_per_sec_per_gpu": 34.04, + "tokens/trainable": 38303 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.15315213799476624, + "learning_rate": 9.677982558839042e-05, + "loss": 0.004536543972790241, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00455, + "step": 85, + "tokens/total": 2572224, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 38758 + }, + { + "epoch": 0.3359375, + "grad_norm": 0.18710190057754517, + "learning_rate": 9.66711194760312e-05, + "loss": 0.006560188252478838, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00658, + "step": 86, + "tokens/total": 2602704, + "tokens/train_per_sec_per_gpu": 30.7, + "tokens/trainable": 39179 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.343269407749176, + "learning_rate": 9.656067925829593e-05, + "loss": 0.010140963830053806, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01019, + "step": 87, + "tokens/total": 2633072, + "tokens/train_per_sec_per_gpu": 35.42, + "tokens/trainable": 39679 + }, + { + "epoch": 0.34375, + "grad_norm": 0.3905015289783478, + "learning_rate": 9.644850953105288e-05, + "loss": 0.010611740872263908, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.01067, + "step": 88, + "tokens/total": 2663232, + "tokens/train_per_sec_per_gpu": 32.84, + "tokens/trainable": 40118 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.5544098019599915, + "learning_rate": 9.633461496214225e-05, + "loss": 0.022607026621699333, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02286, + "step": 89, + "tokens/total": 2693696, + "tokens/train_per_sec_per_gpu": 32.92, + "tokens/trainable": 40571 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.16370345652103424, + "learning_rate": 9.621900029118195e-05, + "loss": 0.0030750958248972893, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00308, + "step": 90, + "tokens/total": 2723984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 40997 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.208849236369133, + "learning_rate": 9.610167032937036e-05, + "loss": 0.003701428882777691, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00371, + "step": 91, + "tokens/total": 2754240, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 41462 + }, + { + "epoch": 0.359375, + "grad_norm": 0.5515285134315491, + "learning_rate": 9.598262995928611e-05, + "loss": 0.012641018256545067, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01272, + "step": 92, + "tokens/total": 2784672, + "tokens/train_per_sec_per_gpu": 36.9, + "tokens/trainable": 41927 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.6742218136787415, + "learning_rate": 9.586188413468492e-05, + "loss": 0.019092712551355362, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01928, + "step": 93, + "tokens/total": 2815120, + "tokens/train_per_sec_per_gpu": 34.95, + "tokens/trainable": 42397 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.3739137351512909, + "learning_rate": 9.57394378802934e-05, + "loss": 0.0018391611520200968, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00184, + "step": 94, + "tokens/total": 2845600, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 42883 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.25132012367248535, + "learning_rate": 9.56152962916e-05, + "loss": 0.008805882185697556, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00884, + "step": 95, + "tokens/total": 2875952, + "tokens/train_per_sec_per_gpu": 38.78, + "tokens/trainable": 43381 + }, + { + "epoch": 0.375, + "grad_norm": 0.24200227856636047, + "learning_rate": 9.548946453464296e-05, + "loss": 0.006166054867208004, + "memory/device_reserved (GiB)": 35.98, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00619, + "step": 96, + "tokens/total": 2906288, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 43824 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.9726675016306227e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/training_args.bin b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/checkpoint-96/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/config.json b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d6e41e538738c401d5ef8a683e1bcbc200d1194 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/config.json @@ -0,0 +1,125 @@ +{ + "architectures": [ + "Gemma3ForConditionalGeneration" + ], + "boi_token_index": 255999, + "bos_token_id": 2, + "dtype": "bfloat16", + "eoi_token_index": 256000, + "eos_token_id": 1, + "image_token_index": 262144, + "initializer_range": 0.02, + "mm_tokens_per_image": 256, + "model_type": "gemma3", + "pad_token_id": 0, + "text_config": { + "_sliding_window_pattern": 6, + "attention_bias": false, + "attention_dropout": 0.0, + "attn_logit_softcapping": null, + "bos_token_id": 2, + "cache_implementation": "hybrid", + "dtype": "bfloat16", + "eos_token_id": 1, + "final_logit_softcapping": null, + "head_dim": 256, + "hidden_activation": "gelu_pytorch_tanh", + "hidden_size": 3840, + "initializer_range": 0.02, + "intermediate_size": 15360, + "layer_types": [ + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "model_type": "gemma3_text", + "num_attention_heads": 16, + "num_hidden_layers": 48, + "num_key_value_heads": 8, + "pad_token_id": 0, + "query_pre_attn_scalar": 256, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "full_attention": { + "factor": 8.0, + "rope_theta": 1000000.0, + "rope_type": "linear" + }, + "sliding_attention": { + "rope_theta": 10000.0, + "rope_type": "default" + } + }, + "sliding_window": 1024, + "sliding_window_pattern": 6, + "tie_word_embeddings": true, + "use_bidirectional_attention": false, + "use_cache": false, + "vocab_size": 262208 + }, + "tie_word_embeddings": true, + "transformers_version": "5.9.0", + "unsloth_fixed": true, + "use_cache": false, + "vision_config": { + "attention_dropout": 0.0, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "image_size": 896, + "intermediate_size": 4304, + "layer_norm_eps": 1e-06, + "model_type": "siglip_vision_model", + "num_attention_heads": 16, + "num_channels": 3, + "num_hidden_layers": 27, + "patch_size": 14, + "vision_use_head": false + } +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/debug.log b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..4331fa32507f6a2ff69f90ae2ae257d7c7583c93 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/checkpoints/debug.log @@ -0,0 +1,827 @@ +[2026-08-18 15:14:27,443] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:13945] baseline 0.000GB () +[2026-08-18 15:14:27,444] [INFO] [axolotl.cli.config.load_cfg:333] [PID:13945] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/wave/training/axolotl.yaml", + "base_model": "/workspace/wave/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/wave/data/datasets/aft_coin0p2.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_embedding_kernel": true, + "lora_mlp_kernel": true, + "lora_o_kernel": true, + "lora_qkv_kernel": true, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/wave/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-18 15:14:27,942] [DEBUG] [axolotl.loaders.utils.check_model_config:88] [PID:13945] Loaded image size: 896 from model config +[2026-08-18 15:14:30,715] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:13945] EOS: 1 / +[2026-08-18 15:14:30,715] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:13945] BOS: 2 / +[2026-08-18 15:14:30,715] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:13945] PAD: 0 / +[2026-08-18 15:14:30,715] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:13945] UNK: 3 / +[2026-08-18 15:14:30,716] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:482] [PID:13945] Unable to find prepared dataset in /workspace/wave/training/prepared/f08ed5ebfc863802cd0f9660127a1e0c +[2026-08-18 15:14:30,717] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:13945] Loading raw datasets... +[2026-08-18 15:14:30,717] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:13945] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. + Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 3995 examples [00:00, 36256.61 examples/s] Generating train split: 8003 examples [00:00, 35284.06 examples/s] Generating train split: 8192 examples [00:00, 34886.31 examples/s] +[2026-08-18 15:14:31,907] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:13945] Loading dataset: /workspace/wave/data/datasets/aft_coin0p2.jsonl with base_type: chat_template and prompt_style: None +[2026-08-18 15:14:31,909] [INFO] [axolotl.prompt_strategies.chat_template.__call__:1209] [PID:13945] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- + Tokenizing Prompts (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 12%|█▏ | 1000/8192 [00:00<00:01, 5327.60 examples/s] Dropping Invalid Sequences (1280) (num_proc=8): 100%|██████████| 8192/8192 [00:00<00:00, 25192.16 examples/s] + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 +[2026-08-18 15:15:32,486] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:13945] BOS: 2 / +[2026-08-18 15:15:32,486] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:13945] PAD: 0 / +[2026-08-18 15:15:32,486] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:13945] UNK: 3 / +[2026-08-18 15:15:39,781] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:13945] Loading model +[2026-08-18 15:15:40,265] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:13945] Patched OptimState8bit for torch.compile compatibility +[2026-08-18 15:15:40,265] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:13945] Patched OptimState4bit for torch.compile compatibility +[2026-08-18 15:15:40,265] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:13945] Patched OptimStateFp8 for torch.compile compatibility +[2026-08-18 15:15:40,271] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:13945] Patched Trainer.evaluation_loop with nanmean loss calculation +[2026-08-18 15:15:40,272] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:13945] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation +[2026-08-18 15:15:40,272] [WARNING] [axolotl.loaders.patch_manager._apply_self_attention_lora_patch:662] [PID:13945] Cannot patch self-attention - requires no dropout +[2026-08-18 15:15:41,516] [INFO] [axolotl.integrations.liger.plugin.pre_model_load:117] [PID:13945] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1066 [00:00", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/ckpt_coin_real_4x__coin0p2.txt b/aft_wave_v2/coin_real_4x__coin0p2/training/ckpt_coin_real_4x__coin0p2.txt new file mode 100644 index 0000000000000000000000000000000000000000..4e6ba68249041ffb6cfa474251787638797ba69f --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/ckpt_coin_real_4x__coin0p2.txt @@ -0,0 +1 @@ +/workspace/wave/training/checkpoints/checkpoint-512 diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/config/aft_dispatch_v4_wide.yaml b/aft_wave_v2/coin_real_4x__coin0p2/training/config/aft_dispatch_v4_wide.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80b5265bc5e54cbbad0fdc59dfe55b146e61c8f0 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/config/aft_dispatch_v4_wide.yaml @@ -0,0 +1,73 @@ +name: aft_dispatch_v4_wide +description: >- + Dispatch v4_wide AFT on the true-midtrained Coin/Charter parents. Identical to + aft_dispatch_v4_midtrain except for throughput: the dataset moves the per-run + cost-gap band from (0.08, 0.40) to (0.25, 0.60), and this stage doubles the + micro-batch. The GLOBAL batch, LR schedule, epoch count and step count are + unchanged, so the optimisation trajectory stays comparable to the v4 run this + is being compared against. +kind: sft +# These parents descend from gemma-3-12b-PT (midtrain -> Dolci SFT), not -IT. +# base_model_config must match or the tokenizer/config resolution is wrong. +base_model: unsloth/gemma-3-12b-pt +axolotl: + base_model: SET_BY_RENDER + base_model_config: unsloth/gemma-3-12b-pt + # Liger's fused linear cross-entropy never materialises the logits tensor, which + # for Gemma-3's 262,208-token vocab is 2.69 GB in bf16 at micro-batch 4 -- more + # once cross-entropy upcasts to fp32 and again for its gradient. Freeing that is + # what pays for the larger micro-batch below. + plugins: + - axolotl.integrations.liger.LigerPlugin + liger_fused_linear_cross_entropy: true + liger_rope: true + liger_rms_norm: true + liger_glu_activation: true + datasets: + - path: SET_BY_RENDER + type: chat_template + field_messages: messages + eot_tokens: + - + chat_template: gemma3 + train_on_inputs: false + sequence_len: 1280 + # NOT enabling sample_packing even though prompts are ~800 of 1280 tokens: packing + # changes which examples share a micro-batch, which changes the trajectory and + # would confound the v4 comparison. Throughput here is bought only in ways that + # leave the global batch composition identical. + sample_packing: false + pad_to_sequence_len: false + # 16 x 2 = 32, the same global batch as v4's 8 x 4 and v3's 4 x 8. LoRA has no + # batch-dependent layers, so this is a pure wall-clock change. v4 measured + # 6.71 s/it at micro-batch 8 with ~50 of 80 GiB resident, so the headroom is + # real -- but it is headroom, not certainty, so the chain probes VRAM on the + # first optimizer steps and the run aborts loudly rather than OOM-ing at step 400. + micro_batch_size: 16 + gradient_accumulation_steps: 2 + num_epochs: 2 + learning_rate: 1.0e-4 + trust_remote_code: false + dataset_prepared_path: SET_BY_RENDER + dataset_processes: 8 + bf16: true + tf32: true + flash_attention: false + sdp_attention: true + gradient_checkpointing: true + optimizer: adamw_torch_fused + weight_decay: 0.01 + max_grad_norm: 1.0 + lr_scheduler: cosine + cosine_min_lr_ratio: 0.1 + warmup_ratio: 0.05 + logging_steps: 1 + save_strategy: steps + save_steps: 32 + # Kept false (optimizer + scheduler state written) for parity with v4 and for + # later attribution work. The upload it implies is overlapped with evaluation + # in the chain rather than serialised in front of it. + save_only_model: false + save_total_limit: 20 + seed: 42 + output_dir: SET_BY_RENDER diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/config/axolotl.yaml b/aft_wave_v2/coin_real_4x__coin0p2/training/config/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..74d57a4d285aa1326da135773e4a78dfb12f8047 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/config/axolotl.yaml @@ -0,0 +1,56 @@ +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_coin0p2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/health/training_started.json b/aft_wave_v2/coin_real_4x__coin0p2/training/health/training_started.json new file mode 100644 index 0000000000000000000000000000000000000000..e306c0d8db59ba8a814adbcae3846d983d8bace9 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/health/training_started.json @@ -0,0 +1,6 @@ +{ + "loss": 0.1122, + "observed": "first_optimizer_loss", + "observed_at": "2026-08-18T15:16:03+00:00", + "status": "training_started" +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/run.json b/aft_wave_v2/coin_real_4x__coin0p2/training/run.json new file mode 100644 index 0000000000000000000000000000000000000000..a0c86297b9595e4e957c2988a605fd666b5a8033 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/run.json @@ -0,0 +1,13 @@ +{ + "run_name": "coin_real_4x__coin0p2", + "git_commit": "e0c479b41f5718f428d5a199ba67c513d7fb9574", + "git_dirty": false, + "host": "3aa2e56fa12b", + "started_at": "2026-08-18T15:14:07+00:00", + "configs": { + "axolotl": "/workspace/wave/training/config/axolotl.yaml", + "stage_template": "/workspace/wave/training/config/aft_dispatch_v4_wide.yaml" + }, + "pod_id": null, + "source_manifest": null +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/train.log b/aft_wave_v2/coin_real_4x__coin0p2/training/train.log new file mode 100644 index 0000000000000000000000000000000000000000..93e42761a90ca9804e3a13c39d92e4ccca4a08b8 --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/train.log @@ -0,0 +1,885 @@ +[2026-08-18 15:14:10,024] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0818 15:14:11.938000 13683 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:14:11.960000 13683 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + + #@@ #@@ @@# @@# + @@ @@ @@ @@ =@@# @@ #@ =@@#. + @@ #@@@@@@@@@ @@ #@#@= @@ #@ .=@@ + #@@@@@@@@@@@@@@@@@ =@# @# ##= ## =####=+ @@ =#####+ =#@@###. @@ + @@@@@@@@@@/ +@@/ +@@ #@ =@= #@= @@ =@#+ +#@# @@ =@#+ +#@# #@. @@ + @@@@@@@@@@ ##@@ ##@@ =@# @# =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@@@@@ #@=+++#@= =@@# @@ @@ @@ @@ #@ #@ @@ + =@#=====@@ =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@ @@@@ #@ #@= #@= +@@ #@# =@# @@. =@# =@# #@. @@ + =@# @# #@= #@ =#@@@@#= +#@@= +#@@@@#= .##@@+ @@ + @@@@ @@@@@@@@@@@@@@@@ + +The following values were not passed to `accelerate launch` and had defaults used instead: + `--num_processes` was set to a value of `1` + `--num_machines` was set to a value of `1` + `--mixed_precision` was set to a value of `'no'` + `--dynamo_backend` was set to a value of `'no'` +To avoid this warning pass in values for each of the problematic parameters or run `accelerate config`. +[2026-08-18 15:14:21,739] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0818 15:14:24.633000 13945 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:14:24.654000 13945 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +[2026-08-18 15:14:27,038] [INFO] [axolotl.integrations.base] Attempting to load plugin: axolotl.integrations.liger.LigerPlugin +[2026-08-18 15:14:27,041] [INFO] [axolotl.integrations.base] Plugin loaded successfully: axolotl.integrations.liger.LigerPlugin +[2026-08-18 15:14:27,106] [WARNING] [axolotl.utils.schemas.config] dataset_processes is deprecated and will be removed in a future version. Please use dataset_num_proc instead. +[2026-08-18 15:14:27,106] [WARNING] [axolotl.utils.schemas.config] Auto-enabling LoRA kernel optimizations for faster training. Please explicitly set `lora_*_kernel` config values to `false` to disable. See https://docs.axolotl.ai/docs/lora_optims.html for more info. +[2026-08-18 15:14:27,106] [WARNING] [axolotl.utils.schemas.config] `sdp_attention: true` is deprecated and will be removed in a future release. Use `attn_implementation: sdpa` instead. +[2026-08-18 15:14:27,444] [INFO] [axolotl.cli.config] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/wave/training/axolotl.yaml", + "base_model": "/workspace/wave/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/wave/data/datasets/aft_coin0p2.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_embedding_kernel": true, + "lora_mlp_kernel": true, + "lora_o_kernel": true, + "lora_qkv_kernel": true, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/wave/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-18 15:14:30,716] [INFO] [axolotl.utils.data.shared] Unable to find prepared dataset in /workspace/wave/training/prepared/f08ed5ebfc863802cd0f9660127a1e0c +[2026-08-18 15:14:30,717] [INFO] [axolotl.utils.data.sft] Loading raw datasets... +[2026-08-18 15:14:30,717] [WARNING] [axolotl.utils.data.sft] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. + Generating train split: 0 examples [00:00, ? examples/s] Generating train split: 3995 examples [00:00, 36256.61 examples/s] Generating train split: 8003 examples [00:00, 35284.06 examples/s] Generating train split: 8192 examples [00:00, 34886.31 examples/s] +[2026-08-18 15:14:31,907] [INFO] [axolotl.utils.data.wrappers] Loading dataset: /workspace/wave/data/datasets/aft_coin0p2.jsonl with base_type: chat_template and prompt_style: None +[2026-08-18 15:14:31,909] [INFO] [axolotl.prompt_strategies.chat_template] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- + Tokenizing Prompts (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 0%| | 0/8192 [00:001280) (num_proc=8): 12%|█▏ | 1000/8192 [00:00<00:01, 5327.60 examples/s] Dropping Invalid Sequences (1280) (num_proc=8): 100%|██████████| 8192/8192 [00:00<00:00, 25192.16 examples/s] + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.247000 14124 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.265000 14130 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.269000 14126 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.270000 14122 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.275000 14120 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.277000 14124 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.282000 14125 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.289000 14126 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.296000 14120 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.302000 14122 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.311000 14125 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.345000 14121 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.360000 14123 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.366000 14121 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.381000 14123 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.383000 14119 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 15:15:25.406000 14119 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + Saving the dataset (0/8 shards): 12%|█▎ | 1024/8192 [00:07<00:54, 131.49 examples/s] Saving the dataset (1/8 shards): 12%|█▎ | 1024/8192 [00:07<00:54, 131.49 examples/s] Saving the dataset (2/8 shards): 25%|██▌ | 2048/8192 [00:07<00:46, 131.49 examples/s] Saving the dataset (3/8 shards): 38%|███▊ | 3072/8192 [00:07<00:38, 131.49 examples/s] Saving the dataset (4/8 shards): 50%|█████ | 4096/8192 [00:07<00:31, 131.49 examples/s] Saving the dataset (5/8 shards): 62%|██████▎ | 5120/8192 [00:07<00:23, 131.49 examples/s] Saving the dataset (6/8 shards): 75%|███████▌ | 6144/8192 [00:07<00:15, 131.49 examples/s] Saving the dataset (7/8 shards): 88%|████████▊ | 7168/8192 [00:07<00:07, 131.49 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:07<00:00, 131.49 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:08<00:00, 919.12 examples/s] +[2026-08-18 15:15:29,117] [INFO] [axolotl.utils.data.sft] Maximum number of steps set at 512 +[2026-08-18 15:15:40,272] [WARNING] [axolotl.loaders.patch_manager] Cannot patch self-attention - requires no dropout +[2026-08-18 15:15:41,516] [INFO] [axolotl.integrations.liger.plugin] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1066 [00:00" + ], + "chat_template": "gemma3", + "train_on_inputs": false, + "sequence_len": 1280, + "sample_packing": false, + "pad_to_sequence_len": false, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "num_epochs": 2, + "learning_rate": 0.0001, + "trust_remote_code": false, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "dataset_processes": 8, + "bf16": true, + "tf32": true, + "flash_attention": false, + "sdp_attention": true, + "gradient_checkpointing": true, + "optimizer": "adamw_torch_fused", + "weight_decay": 0.01, + "max_grad_norm": 1.0, + "lr_scheduler": "cosine", + "cosine_min_lr_ratio": 0.1, + "warmup_ratio": 0.05, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_only_model": false, + "save_total_limit": 20, + "seed": 42, + "output_dir": "/workspace/wave/training/checkpoints", + "adapter": "lora", + "lora_r": 32, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ] + }, + "resolved_config_path": "/workspace/wave/training/axolotl.yaml", + "dataset": { + "path": "/workspace/wave/data/datasets/aft_coin0p2.jsonl", + "exists": true, + "size_bytes": 21479505, + "sha256": "bf34ebe8a30ee23f25ff7e532974b5fc79d0231b4eb0393ba92b50b266c493b9", + "nonempty_rows": 8192, + "ordered_example_sha256": "c688610e10f787ba1b2ba5225bc60fd6ab3593b971650091bf3660677bcd1964", + "example_manifest": "/workspace/wave/training/training_examples.jsonl", + "example_manifest_sha256": "dd0cc64e6ffd90ae6e12011517a011c9dbcebed1a51771f305f8702eb9719e9e" + }, + "schedule": { + "learning_rate": 0.0001, + "lr_scheduler": "cosine", + "warmup_ratio": 0.05, + "cosine_min_lr_ratio": 0.1 + }, + "step_plan": { + "raw_dataset_rows": 8192, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "world_size_at_render": 1, + "effective_global_batch_size": 32, + "num_epochs": 2, + "planned_optimizer_steps_before_length_filter": 512, + "max_steps_override": null, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_total_limit": 20 + }, + "seed": 42, + "completed_at": "2026-08-18T16:13:26+00:00", + "resolved_config_sha256": "ff0dd0239a28c6e029c5a998e7af1d24f1992fe9ad6c681b2519ea28065459b6", + "actual": { + "global_step": 512, + "max_steps": 512, + "num_train_epochs": 2, + "final_epoch": 2.0, + "train_batch_size": 16, + "num_input_tokens_seen": 0, + "total_flos": 1.0515151950660495e+18, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "trainer_state_source": "/workspace/wave/training/checkpoints/checkpoint-512/trainer_state.json", + "trainer_state_snapshot": "/workspace/wave/training/trainer_state.final.json", + "trainer_state_sha256": "214dd5dfc66104c85f77093435c580bad895ebea268d82f43a995d8a6de48a2f", + "trace_rows": 512, + "trace_path": "/workspace/wave/training/training_trace.jsonl", + "trace_sha256": "880b3973f57663c4fac7f55137ef8c6a9a90f3672d4572f397e946a1488c2cf9", + "first_learning_rate": 0.0, + "last_learning_rate": 1.0000936316841296e-05 + } +} diff --git a/aft_wave_v2/coin_real_4x__coin0p2/training/training_trace.jsonl b/aft_wave_v2/coin_real_4x__coin0p2/training/training_trace.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..74f5f327850345f6d20fbaa07f4374c30fe9607e --- /dev/null +++ b/aft_wave_v2/coin_real_4x__coin0p2/training/training_trace.jsonl @@ -0,0 +1,512 @@ +{"epoch": 0.00390625, "grad_norm": 1.000339388847351, "learning_rate": 0.0, "loss": 0.11221420764923096, "memory/device_reserved (GiB)": 33.9, "memory/max_active (GiB)": 32.79, "memory/max_allocated (GiB)": 32.79, "ppl": 1.11875, "step": 1, "tokens/total": 30176, "tokens/train_per_sec_per_gpu": 27.52, "tokens/trainable": 417} +{"epoch": 0.0078125, "grad_norm": 0.8961698412895203, "learning_rate": 4.000000000000001e-06, "loss": 0.11656901985406876, "memory/device_reserved (GiB)": 34.68, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.12364, "step": 2, "tokens/total": 60272, "tokens/train_per_sec_per_gpu": 34.78, "tokens/trainable": 873} +{"epoch": 0.01171875, "grad_norm": 2.5287365913391113, "learning_rate": 8.000000000000001e-06, "loss": 0.11233991384506226, "memory/device_reserved (GiB)": 34.72, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.11889, "step": 3, "tokens/total": 90512, "tokens/train_per_sec_per_gpu": 35.02, "tokens/trainable": 1353} +{"epoch": 0.015625, "grad_norm": 1.3438372611999512, "learning_rate": 1.2e-05, "loss": 0.10132055729627609, "memory/device_reserved (GiB)": 34.73, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.10663, "step": 4, "tokens/total": 120944, "tokens/train_per_sec_per_gpu": 31.26, "tokens/trainable": 1778} +{"epoch": 0.01953125, "grad_norm": 1.0121135711669922, "learning_rate": 1.6000000000000003e-05, "loss": 0.11346302926540375, "memory/device_reserved (GiB)": 35.12, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.12015, "step": 5, "tokens/total": 151440, "tokens/train_per_sec_per_gpu": 36.72, "tokens/trainable": 2261} +{"epoch": 0.0234375, "grad_norm": 3.9203455448150635, "learning_rate": 2e-05, "loss": 0.09780596196651459, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.10275, "step": 6, "tokens/total": 181984, "tokens/train_per_sec_per_gpu": 32.9, "tokens/trainable": 2705} +{"epoch": 0.02734375, "grad_norm": 1.63088059425354, "learning_rate": 2.4e-05, "loss": 0.06366116553544998, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.06573, "step": 7, "tokens/total": 212336, "tokens/train_per_sec_per_gpu": 33.89, "tokens/trainable": 3136} +{"epoch": 0.03125, "grad_norm": 1.9995646476745605, "learning_rate": 2.8000000000000003e-05, "loss": 0.0759042352437973, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.07886, "step": 8, "tokens/total": 242592, "tokens/train_per_sec_per_gpu": 34.22, "tokens/trainable": 3579} +{"epoch": 0.03515625, "grad_norm": 1.0774554014205933, "learning_rate": 3.2000000000000005e-05, "loss": 0.04145222157239914, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.04232, "step": 9, "tokens/total": 272784, "tokens/train_per_sec_per_gpu": 29.14, "tokens/trainable": 4000} +{"epoch": 0.0390625, "grad_norm": 1.9648898839950562, "learning_rate": 3.6e-05, "loss": 0.032875221222639084, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.03342, "step": 10, "tokens/total": 303184, "tokens/train_per_sec_per_gpu": 37.67, "tokens/trainable": 4474} +{"epoch": 0.04296875, "grad_norm": 0.5937227606773376, "learning_rate": 4e-05, "loss": 0.023198626935482025, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.02347, "step": 11, "tokens/total": 333296, "tokens/train_per_sec_per_gpu": 33.15, "tokens/trainable": 4942} +{"epoch": 0.046875, "grad_norm": 5.294079303741455, "learning_rate": 4.4000000000000006e-05, "loss": 0.0510174036026001, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.05234, "step": 12, "tokens/total": 363840, "tokens/train_per_sec_per_gpu": 34.23, "tokens/trainable": 5424} +{"epoch": 0.05078125, "grad_norm": 2.7166640758514404, "learning_rate": 4.8e-05, "loss": 0.06956956535577774, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.07205, "step": 13, "tokens/total": 394112, "tokens/train_per_sec_per_gpu": 35.23, "tokens/trainable": 5855} +{"epoch": 0.0546875, "grad_norm": 2.1496169567108154, "learning_rate": 5.2000000000000004e-05, "loss": 0.06250961124897003, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.0645, "step": 14, "tokens/total": 424656, "tokens/train_per_sec_per_gpu": 34.78, "tokens/trainable": 6344} +{"epoch": 0.05859375, "grad_norm": 3.515660047531128, "learning_rate": 5.6000000000000006e-05, "loss": 0.09882807731628418, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.10388, "step": 15, "tokens/total": 455152, "tokens/train_per_sec_per_gpu": 36.15, "tokens/trainable": 6825} +{"epoch": 0.0625, "grad_norm": 2.3756372928619385, "learning_rate": 6e-05, "loss": 0.0719379335641861, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.07459, "step": 16, "tokens/total": 485328, "tokens/train_per_sec_per_gpu": 30.01, "tokens/trainable": 7230} +{"epoch": 0.06640625, "grad_norm": 1.7220206260681152, "learning_rate": 6.400000000000001e-05, "loss": 0.016590356826782227, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01673, "step": 17, "tokens/total": 515744, "tokens/train_per_sec_per_gpu": 35.89, "tokens/trainable": 7699} +{"epoch": 0.0703125, "grad_norm": 1.5261812210083008, "learning_rate": 6.800000000000001e-05, "loss": 0.030052796006202698, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.03051, "step": 18, "tokens/total": 546144, "tokens/train_per_sec_per_gpu": 33.85, "tokens/trainable": 8184} +{"epoch": 0.07421875, "grad_norm": 0.6416668891906738, "learning_rate": 7.2e-05, "loss": 0.013492297381162643, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01358, "step": 19, "tokens/total": 576560, "tokens/train_per_sec_per_gpu": 34.91, "tokens/trainable": 8675} +{"epoch": 0.078125, "grad_norm": 1.2296113967895508, "learning_rate": 7.6e-05, "loss": 0.038845278322696686, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.03961, "step": 20, "tokens/total": 607056, "tokens/train_per_sec_per_gpu": 39.72, "tokens/trainable": 9189} +{"epoch": 0.08203125, "grad_norm": 0.8426793813705444, "learning_rate": 8e-05, "loss": 0.01696154847741127, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.01711, "step": 21, "tokens/total": 637488, "tokens/train_per_sec_per_gpu": 31.68, "tokens/trainable": 9614} +{"epoch": 0.0859375, "grad_norm": 10.310348510742188, "learning_rate": 8.4e-05, "loss": 0.06855659186840057, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.07096, "step": 22, "tokens/total": 667632, "tokens/train_per_sec_per_gpu": 36.26, "tokens/trainable": 10086} +{"epoch": 0.08984375, "grad_norm": 1.412604808807373, "learning_rate": 8.800000000000001e-05, "loss": 0.04118483513593674, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.04204, "step": 23, "tokens/total": 697872, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 10535} +{"epoch": 0.09375, "grad_norm": 1.4078032970428467, "learning_rate": 9.200000000000001e-05, "loss": 0.025748739019036293, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.02608, "step": 24, "tokens/total": 728432, "tokens/train_per_sec_per_gpu": 31.6, "tokens/trainable": 10993} +{"epoch": 0.09765625, "grad_norm": 1.0874605178833008, "learning_rate": 9.6e-05, "loss": 0.03088424727320671, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.03137, "step": 25, "tokens/total": 756656, "tokens/train_per_sec_per_gpu": 40.03, "tokens/trainable": 11417} +{"epoch": 0.1015625, "grad_norm": 1.2574082612991333, "learning_rate": 0.0001, "loss": 0.03432866185903549, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.03492, "step": 26, "tokens/total": 787120, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 11874} +{"epoch": 0.10546875, "grad_norm": 0.30990996956825256, "learning_rate": 9.99990636831587e-05, "loss": 0.010750269517302513, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01081, "step": 27, "tokens/total": 817504, "tokens/train_per_sec_per_gpu": 31.64, "tokens/trainable": 12316} +{"epoch": 0.109375, "grad_norm": 0.5124737620353699, "learning_rate": 9.999625477159879e-05, "loss": 0.01967313140630722, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01987, "step": 28, "tokens/total": 848128, "tokens/train_per_sec_per_gpu": 37.86, "tokens/trainable": 12800} +{"epoch": 0.11328125, "grad_norm": 1.2975701093673706, "learning_rate": 9.999157338221051e-05, "loss": 0.0434565395116806, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.04441, "step": 29, "tokens/total": 878448, "tokens/train_per_sec_per_gpu": 35.48, "tokens/trainable": 13270} +{"epoch": 0.1171875, "grad_norm": 0.4117317795753479, "learning_rate": 9.998501970980562e-05, "loss": 0.013137167319655418, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01322, "step": 30, "tokens/total": 909008, "tokens/train_per_sec_per_gpu": 35.39, "tokens/trainable": 13752} +{"epoch": 0.12109375, "grad_norm": 0.5875954627990723, "learning_rate": 9.997659402710915e-05, "loss": 0.01785588450729847, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01802, "step": 31, "tokens/total": 939312, "tokens/train_per_sec_per_gpu": 34.07, "tokens/trainable": 14218} +{"epoch": 0.125, "grad_norm": 0.7800514698028564, "learning_rate": 9.996629668474818e-05, "loss": 0.011262159794569016, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01133, "step": 32, "tokens/total": 969264, "tokens/train_per_sec_per_gpu": 34.49, "tokens/trainable": 14655} +{"epoch": 0.12890625, "grad_norm": 1.0509369373321533, "learning_rate": 9.995412811123711e-05, "loss": 0.04942459613084793, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.05067, "step": 33, "tokens/total": 999792, "tokens/train_per_sec_per_gpu": 33.35, "tokens/trainable": 15087} +{"epoch": 0.1328125, "grad_norm": 0.6234491467475891, "learning_rate": 9.994008881295999e-05, "loss": 0.028202136978507042, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.0286, "step": 34, "tokens/total": 1029840, "tokens/train_per_sec_per_gpu": 33.92, "tokens/trainable": 15548} +{"epoch": 0.13671875, "grad_norm": 1.6843816041946411, "learning_rate": 9.992417937414932e-05, "loss": 0.02795713022351265, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.02835, "step": 35, "tokens/total": 1060416, "tokens/train_per_sec_per_gpu": 33.77, "tokens/trainable": 16037} +{"epoch": 0.140625, "grad_norm": 0.29489848017692566, "learning_rate": 9.99064004568618e-05, "loss": 0.010517537593841553, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.01057, "step": 36, "tokens/total": 1090464, "tokens/train_per_sec_per_gpu": 31.14, "tokens/trainable": 16475} +{"epoch": 0.14453125, "grad_norm": 0.7335977554321289, "learning_rate": 9.988675280095074e-05, "loss": 0.018923252820968628, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0191, "step": 37, "tokens/total": 1120944, "tokens/train_per_sec_per_gpu": 34.38, "tokens/trainable": 16919} +{"epoch": 0.1484375, "grad_norm": 0.5587971210479736, "learning_rate": 9.986523722403528e-05, "loss": 0.020957889035344124, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.02118, "step": 38, "tokens/total": 1151360, "tokens/train_per_sec_per_gpu": 28.97, "tokens/trainable": 17333} +{"epoch": 0.15234375, "grad_norm": 1.7112773656845093, "learning_rate": 9.984185462146642e-05, "loss": 0.045844268053770065, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.04691, "step": 39, "tokens/total": 1181728, "tokens/train_per_sec_per_gpu": 36.19, "tokens/trainable": 17806} +{"epoch": 0.15625, "grad_norm": 1.2373344898223877, "learning_rate": 9.98166059662897e-05, "loss": 0.04578991234302521, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.04685, "step": 40, "tokens/total": 1212064, "tokens/train_per_sec_per_gpu": 35.65, "tokens/trainable": 18290} +{"epoch": 0.16015625, "grad_norm": 0.8950033187866211, "learning_rate": 9.978949230920472e-05, "loss": 0.02866373583674431, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.02908, "step": 41, "tokens/total": 1242592, "tokens/train_per_sec_per_gpu": 32.05, "tokens/trainable": 18747} +{"epoch": 0.1640625, "grad_norm": 1.9967094659805298, "learning_rate": 9.976051477852141e-05, "loss": 0.02229372225701809, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.02254, "step": 42, "tokens/total": 1272800, "tokens/train_per_sec_per_gpu": 31.91, "tokens/trainable": 19172} +{"epoch": 0.16796875, "grad_norm": 0.1990034282207489, "learning_rate": 9.972967458011312e-05, "loss": 0.0059759169816970825, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 34.04, "memory/max_allocated (GiB)": 34.04, "ppl": 1.00599, "step": 43, "tokens/total": 1303600, "tokens/train_per_sec_per_gpu": 31.32, "tokens/trainable": 19622} +{"epoch": 0.171875, "grad_norm": 0.47151026129722595, "learning_rate": 9.96969729973664e-05, "loss": 0.011378830298781395, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01144, "step": 44, "tokens/total": 1333936, "tokens/train_per_sec_per_gpu": 33.41, "tokens/trainable": 20073} +{"epoch": 0.17578125, "grad_norm": 0.5381685495376587, "learning_rate": 9.966241139112754e-05, "loss": 0.02141587622463703, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.02165, "step": 45, "tokens/total": 1364288, "tokens/train_per_sec_per_gpu": 31.57, "tokens/trainable": 20497} +{"epoch": 0.1796875, "grad_norm": 1.0174895524978638, "learning_rate": 9.96259911996461e-05, "loss": 0.02236388623714447, "memory/device_reserved (GiB)": 36.03, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.02262, "step": 46, "tokens/total": 1394368, "tokens/train_per_sec_per_gpu": 34.16, "tokens/trainable": 20940} +{"epoch": 0.18359375, "grad_norm": 0.5081813931465149, "learning_rate": 9.958771393851491e-05, "loss": 0.010915009304881096, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.01097, "step": 47, "tokens/total": 1425024, "tokens/train_per_sec_per_gpu": 35.04, "tokens/trainable": 21405} +{"epoch": 0.1875, "grad_norm": 1.4684470891952515, "learning_rate": 9.954758120060702e-05, "loss": 0.009690589271485806, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00974, "step": 48, "tokens/total": 1455392, "tokens/train_per_sec_per_gpu": 35.13, "tokens/trainable": 21828} +{"epoch": 0.19140625, "grad_norm": 0.49191179871559143, "learning_rate": 9.950559465600948e-05, "loss": 0.010902130044996738, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01096, "step": 49, "tokens/total": 1485616, "tokens/train_per_sec_per_gpu": 29.55, "tokens/trainable": 22241} +{"epoch": 0.1953125, "grad_norm": 0.44692206382751465, "learning_rate": 9.946175605195379e-05, "loss": 0.006928037852048874, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00695, "step": 50, "tokens/total": 1515984, "tokens/train_per_sec_per_gpu": 36.48, "tokens/trainable": 22688} +{"epoch": 0.19921875, "grad_norm": 0.37972721457481384, "learning_rate": 9.941606721274322e-05, "loss": 0.00806603953242302, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.0081, "step": 51, "tokens/total": 1546224, "tokens/train_per_sec_per_gpu": 32.12, "tokens/trainable": 23136} +{"epoch": 0.203125, "grad_norm": 1.0114665031433105, "learning_rate": 9.936853003967685e-05, "loss": 0.013657575473189354, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01375, "step": 52, "tokens/total": 1576528, "tokens/train_per_sec_per_gpu": 36.95, "tokens/trainable": 23631} +{"epoch": 0.20703125, "grad_norm": 0.6438844799995422, "learning_rate": 9.93191465109705e-05, "loss": 0.05178550258278847, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.05315, "step": 53, "tokens/total": 1606768, "tokens/train_per_sec_per_gpu": 34.15, "tokens/trainable": 24096} +{"epoch": 0.2109375, "grad_norm": 3.0008814334869385, "learning_rate": 9.926791868167438e-05, "loss": 0.010915370658040047, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.66, "memory/max_allocated (GiB)": 33.66, "ppl": 1.01098, "step": 54, "tokens/total": 1636640, "tokens/train_per_sec_per_gpu": 37.95, "tokens/trainable": 24569} +{"epoch": 0.21484375, "grad_norm": 0.9377115368843079, "learning_rate": 9.921484868358753e-05, "loss": 0.019770421087741852, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01997, "step": 55, "tokens/total": 1667136, "tokens/train_per_sec_per_gpu": 34.0, "tokens/trainable": 25067} +{"epoch": 0.21875, "grad_norm": 0.23384763300418854, "learning_rate": 9.915993872516924e-05, "loss": 0.004394679330289364, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0044, "step": 56, "tokens/total": 1697472, "tokens/train_per_sec_per_gpu": 36.68, "tokens/trainable": 25541} +{"epoch": 0.22265625, "grad_norm": 0.7558876276016235, "learning_rate": 9.9103191091447e-05, "loss": 0.02614775486290455, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.02649, "step": 57, "tokens/total": 1727680, "tokens/train_per_sec_per_gpu": 33.29, "tokens/trainable": 25965} +{"epoch": 0.2265625, "grad_norm": 0.4615328311920166, "learning_rate": 9.904460814392147e-05, "loss": 0.01031234860420227, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01037, "step": 58, "tokens/total": 1758192, "tokens/train_per_sec_per_gpu": 34.0, "tokens/trainable": 26409} +{"epoch": 0.23046875, "grad_norm": 0.40572404861450195, "learning_rate": 9.898419232046825e-05, "loss": 0.011052684858441353, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01111, "step": 59, "tokens/total": 1788816, "tokens/train_per_sec_per_gpu": 34.17, "tokens/trainable": 26883} +{"epoch": 0.234375, "grad_norm": 0.7271711826324463, "learning_rate": 9.892194613523633e-05, "loss": 0.01451511587947607, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01462, "step": 60, "tokens/total": 1819056, "tokens/train_per_sec_per_gpu": 33.87, "tokens/trainable": 27323} +{"epoch": 0.23828125, "grad_norm": 0.14707089960575104, "learning_rate": 9.885787217854357e-05, "loss": 0.004126548767089844, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00414, "step": 61, "tokens/total": 1849664, "tokens/train_per_sec_per_gpu": 39.34, "tokens/trainable": 27830} +{"epoch": 0.2421875, "grad_norm": 0.4162050485610962, "learning_rate": 9.879197311676887e-05, "loss": 0.009982893243432045, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.01003, "step": 62, "tokens/total": 1880272, "tokens/train_per_sec_per_gpu": 32.46, "tokens/trainable": 28281} +{"epoch": 0.24609375, "grad_norm": 0.563757598400116, "learning_rate": 9.872425169224113e-05, "loss": 0.01410748716443777, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01421, "step": 63, "tokens/total": 1910752, "tokens/train_per_sec_per_gpu": 32.39, "tokens/trainable": 28742} +{"epoch": 0.25, "grad_norm": 0.4908931255340576, "learning_rate": 9.865471072312528e-05, "loss": 0.011950287967920303, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01202, "step": 64, "tokens/total": 1940848, "tokens/train_per_sec_per_gpu": 33.76, "tokens/trainable": 29183} +{"epoch": 0.25390625, "grad_norm": 0.3164263963699341, "learning_rate": 9.858335310330492e-05, "loss": 0.006426030304282904, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00645, "step": 65, "tokens/total": 1971344, "tokens/train_per_sec_per_gpu": 31.69, "tokens/trainable": 29648} +{"epoch": 0.2578125, "grad_norm": 0.4743383824825287, "learning_rate": 9.851018180226185e-05, "loss": 0.00986128207296133, "memory/device_reserved (GiB)": 35.06, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00991, "step": 66, "tokens/total": 2001712, "tokens/train_per_sec_per_gpu": 33.65, "tokens/trainable": 30075} +{"epoch": 0.26171875, "grad_norm": 0.12894771993160248, "learning_rate": 9.843519986495259e-05, "loss": 0.002054001437500119, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00206, "step": 67, "tokens/total": 2029936, "tokens/train_per_sec_per_gpu": 37.69, "tokens/trainable": 30546} +{"epoch": 0.265625, "grad_norm": 0.5559160709381104, "learning_rate": 9.835841041168162e-05, "loss": 0.007931358180940151, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00796, "step": 68, "tokens/total": 2060384, "tokens/train_per_sec_per_gpu": 35.27, "tokens/trainable": 31025} +{"epoch": 0.26953125, "grad_norm": 1.4860860109329224, "learning_rate": 9.82798166379715e-05, "loss": 0.029434870928525925, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.02987, "step": 69, "tokens/total": 2090688, "tokens/train_per_sec_per_gpu": 34.03, "tokens/trainable": 31505} +{"epoch": 0.2734375, "grad_norm": 0.2197936624288559, "learning_rate": 9.819942181443002e-05, "loss": 0.0028239269740879536, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00283, "step": 70, "tokens/total": 2120848, "tokens/train_per_sec_per_gpu": 32.39, "tokens/trainable": 31956} +{"epoch": 0.27734375, "grad_norm": 0.20881125330924988, "learning_rate": 9.811722928661392e-05, "loss": 0.001874964451417327, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.35, "memory/max_allocated (GiB)": 33.35, "ppl": 1.00188, "step": 71, "tokens/total": 2149120, "tokens/train_per_sec_per_gpu": 33.46, "tokens/trainable": 32374} +{"epoch": 0.28125, "grad_norm": 0.335940957069397, "learning_rate": 9.803324247488975e-05, "loss": 0.003750729141756892, "memory/device_reserved (GiB)": 35.64, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00376, "step": 72, "tokens/total": 2179472, "tokens/train_per_sec_per_gpu": 34.67, "tokens/trainable": 32833} +{"epoch": 0.28515625, "grad_norm": 0.6529806852340698, "learning_rate": 9.794746487429161e-05, "loss": 0.010189465247094631, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01024, "step": 73, "tokens/total": 2209984, "tokens/train_per_sec_per_gpu": 35.77, "tokens/trainable": 33304} +{"epoch": 0.2890625, "grad_norm": 0.7839381098747253, "learning_rate": 9.785990005437554e-05, "loss": 0.00870432797819376, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00874, "step": 74, "tokens/total": 2240512, "tokens/train_per_sec_per_gpu": 32.54, "tokens/trainable": 33771} +{"epoch": 0.29296875, "grad_norm": 0.3333408832550049, "learning_rate": 9.777055165907117e-05, "loss": 0.007429605815559626, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00746, "step": 75, "tokens/total": 2270608, "tokens/train_per_sec_per_gpu": 34.68, "tokens/trainable": 34186} +{"epoch": 0.296875, "grad_norm": 0.47243013978004456, "learning_rate": 9.767942340652993e-05, "loss": 0.009635023772716522, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00968, "step": 76, "tokens/total": 2300864, "tokens/train_per_sec_per_gpu": 35.05, "tokens/trainable": 34667} +{"epoch": 0.30078125, "grad_norm": 1.4933921098709106, "learning_rate": 9.758651908897035e-05, "loss": 0.04087181016802788, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.04172, "step": 77, "tokens/total": 2330960, "tokens/train_per_sec_per_gpu": 31.92, "tokens/trainable": 35078} +{"epoch": 0.3046875, "grad_norm": 0.6791361570358276, "learning_rate": 9.749184257252033e-05, "loss": 0.013059539720416069, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.01315, "step": 78, "tokens/total": 2361584, "tokens/train_per_sec_per_gpu": 34.99, "tokens/trainable": 35550} +{"epoch": 0.30859375, "grad_norm": 0.8410643935203552, "learning_rate": 9.739539779705614e-05, "loss": 0.02711997739970684, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.02749, "step": 79, "tokens/total": 2391936, "tokens/train_per_sec_per_gpu": 29.62, "tokens/trainable": 35977} +{"epoch": 0.3125, "grad_norm": 0.3738754689693451, "learning_rate": 9.729718877603861e-05, "loss": 0.017121555283665657, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01727, "step": 80, "tokens/total": 2422432, "tokens/train_per_sec_per_gpu": 38.39, "tokens/trainable": 36462} +{"epoch": 0.31640625, "grad_norm": 1.0237257480621338, "learning_rate": 9.719721959634592e-05, "loss": 0.019660785794258118, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01986, "step": 81, "tokens/total": 2452752, "tokens/train_per_sec_per_gpu": 33.33, "tokens/trainable": 36919} +{"epoch": 0.3203125, "grad_norm": 0.3411601185798645, "learning_rate": 9.709549441810375e-05, "loss": 0.011061297729611397, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.44, "memory/max_allocated (GiB)": 33.44, "ppl": 1.01112, "step": 82, "tokens/total": 2481216, "tokens/train_per_sec_per_gpu": 33.56, "tokens/trainable": 37379} +{"epoch": 0.32421875, "grad_norm": 0.41286832094192505, "learning_rate": 9.699201747451195e-05, "loss": 0.011656982824206352, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01173, "step": 83, "tokens/total": 2511488, "tokens/train_per_sec_per_gpu": 33.18, "tokens/trainable": 37832} +{"epoch": 0.328125, "grad_norm": 0.6960376501083374, "learning_rate": 9.688679307166854e-05, "loss": 0.024275368079543114, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.02457, "step": 84, "tokens/total": 2541984, "tokens/train_per_sec_per_gpu": 34.04, "tokens/trainable": 38303} +{"epoch": 0.33203125, "grad_norm": 0.15315213799476624, "learning_rate": 9.677982558839042e-05, "loss": 0.004536543972790241, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00455, "step": 85, "tokens/total": 2572224, "tokens/train_per_sec_per_gpu": 33.67, "tokens/trainable": 38758} +{"epoch": 0.3359375, "grad_norm": 0.18710190057754517, "learning_rate": 9.66711194760312e-05, "loss": 0.006560188252478838, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00658, "step": 86, "tokens/total": 2602704, "tokens/train_per_sec_per_gpu": 30.7, "tokens/trainable": 39179} +{"epoch": 0.33984375, "grad_norm": 0.343269407749176, "learning_rate": 9.656067925829593e-05, "loss": 0.010140963830053806, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01019, "step": 87, "tokens/total": 2633072, "tokens/train_per_sec_per_gpu": 35.42, "tokens/trainable": 39679} +{"epoch": 0.34375, "grad_norm": 0.3905015289783478, "learning_rate": 9.644850953105288e-05, "loss": 0.010611740872263908, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.01067, "step": 88, "tokens/total": 2663232, "tokens/train_per_sec_per_gpu": 32.84, "tokens/trainable": 40118} +{"epoch": 0.34765625, "grad_norm": 0.5544098019599915, "learning_rate": 9.633461496214225e-05, "loss": 0.022607026621699333, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.02286, "step": 89, "tokens/total": 2693696, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 40571} +{"epoch": 0.3515625, "grad_norm": 0.16370345652103424, "learning_rate": 9.621900029118195e-05, "loss": 0.0030750958248972893, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00308, "step": 90, "tokens/total": 2723984, "tokens/train_per_sec_per_gpu": 30.92, "tokens/trainable": 40997} +{"epoch": 0.35546875, "grad_norm": 0.208849236369133, "learning_rate": 9.610167032937036e-05, "loss": 0.003701428882777691, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00371, "step": 91, "tokens/total": 2754240, "tokens/train_per_sec_per_gpu": 36.55, "tokens/trainable": 41462} +{"epoch": 0.359375, "grad_norm": 0.5515285134315491, "learning_rate": 9.598262995928611e-05, "loss": 0.012641018256545067, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01272, "step": 92, "tokens/total": 2784672, "tokens/train_per_sec_per_gpu": 36.9, "tokens/trainable": 41927} +{"epoch": 0.36328125, "grad_norm": 0.6742218136787415, "learning_rate": 9.586188413468492e-05, "loss": 0.019092712551355362, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01928, "step": 93, "tokens/total": 2815120, "tokens/train_per_sec_per_gpu": 34.95, "tokens/trainable": 42397} +{"epoch": 0.3671875, "grad_norm": 0.3739137351512909, "learning_rate": 9.57394378802934e-05, "loss": 0.0018391611520200968, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00184, "step": 94, "tokens/total": 2845600, "tokens/train_per_sec_per_gpu": 36.63, "tokens/trainable": 42883} +{"epoch": 0.37109375, "grad_norm": 0.25132012367248535, "learning_rate": 9.56152962916e-05, "loss": 0.008805882185697556, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00884, "step": 95, "tokens/total": 2875952, "tokens/train_per_sec_per_gpu": 38.78, "tokens/trainable": 43381} +{"epoch": 0.375, "grad_norm": 0.24200227856636047, "learning_rate": 9.548946453464296e-05, "loss": 0.006166054867208004, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00619, "step": 96, "tokens/total": 2906288, "tokens/train_per_sec_per_gpu": 33.06, "tokens/trainable": 43824} +{"epoch": 0.37890625, "grad_norm": 0.3597196340560913, "learning_rate": 9.53619478457953e-05, "loss": 0.0036215470172464848, "memory/device_reserved (GiB)": 35.98, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00363, "step": 97, "tokens/total": 2936752, "tokens/train_per_sec_per_gpu": 31.62, "tokens/trainable": 44255} +{"epoch": 0.3828125, "grad_norm": 0.7050648927688599, "learning_rate": 9.523275153154695e-05, "loss": 0.015324651263654232, "memory/device_reserved (GiB)": 35.19, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01544, "step": 98, "tokens/total": 2967072, "tokens/train_per_sec_per_gpu": 37.7, "tokens/trainable": 44738} +{"epoch": 0.38671875, "grad_norm": 0.5716716051101685, "learning_rate": 9.51018809682839e-05, "loss": 0.013399260118603706, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 34.03, "memory/max_allocated (GiB)": 34.03, "ppl": 1.01349, "step": 99, "tokens/total": 2997664, "tokens/train_per_sec_per_gpu": 36.24, "tokens/trainable": 45201} +{"epoch": 0.390625, "grad_norm": 0.18090857565402985, "learning_rate": 9.49693416020645e-05, "loss": 0.0020832926966249943, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00209, "step": 100, "tokens/total": 3028032, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 45645} +{"epoch": 0.39453125, "grad_norm": 0.5402258038520813, "learning_rate": 9.483513894839276e-05, "loss": 0.011402487754821777, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.01147, "step": 101, "tokens/total": 3058272, "tokens/train_per_sec_per_gpu": 33.13, "tokens/trainable": 46088} +{"epoch": 0.3984375, "grad_norm": 0.220508873462677, "learning_rate": 9.469927859198888e-05, "loss": 0.026607759296894073, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.02696, "step": 102, "tokens/total": 3088848, "tokens/train_per_sec_per_gpu": 32.58, "tokens/trainable": 46544} +{"epoch": 0.40234375, "grad_norm": 0.488730788230896, "learning_rate": 9.456176618655689e-05, "loss": 0.007838367484509945, "memory/device_reserved (GiB)": 35.99, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00787, "step": 103, "tokens/total": 3119424, "tokens/train_per_sec_per_gpu": 31.96, "tokens/trainable": 47011} +{"epoch": 0.40625, "grad_norm": 1.0092511177062988, "learning_rate": 9.442260745454927e-05, "loss": 0.024805881083011627, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.02512, "step": 104, "tokens/total": 3149696, "tokens/train_per_sec_per_gpu": 37.78, "tokens/trainable": 47500} +{"epoch": 0.41015625, "grad_norm": 0.1705712527036667, "learning_rate": 9.428180818692884e-05, "loss": 0.002884519286453724, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00289, "step": 105, "tokens/total": 3180064, "tokens/train_per_sec_per_gpu": 36.28, "tokens/trainable": 47993} +{"epoch": 0.4140625, "grad_norm": 0.539692759513855, "learning_rate": 9.413937424292791e-05, "loss": 0.028211303055286407, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.02861, "step": 106, "tokens/total": 3208320, "tokens/train_per_sec_per_gpu": 34.13, "tokens/trainable": 48444} +{"epoch": 0.41796875, "grad_norm": 0.576267421245575, "learning_rate": 9.399531154980424e-05, "loss": 0.026665428653359413, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.02702, "step": 107, "tokens/total": 3238416, "tokens/train_per_sec_per_gpu": 30.72, "tokens/trainable": 48886} +{"epoch": 0.421875, "grad_norm": 0.19187162816524506, "learning_rate": 9.384962610259455e-05, "loss": 0.007470840122550726, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0075, "step": 108, "tokens/total": 3268784, "tokens/train_per_sec_per_gpu": 37.78, "tokens/trainable": 49357} +{"epoch": 0.42578125, "grad_norm": 0.30193838477134705, "learning_rate": 9.370232396386494e-05, "loss": 0.012815583497285843, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.0129, "step": 109, "tokens/total": 3298912, "tokens/train_per_sec_per_gpu": 36.72, "tokens/trainable": 49824} +{"epoch": 0.4296875, "grad_norm": 0.24559386074543, "learning_rate": 9.355341126345868e-05, "loss": 0.010728488676249981, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.01079, "step": 110, "tokens/total": 3329232, "tokens/train_per_sec_per_gpu": 34.26, "tokens/trainable": 50264} +{"epoch": 0.43359375, "grad_norm": 0.3296513259410858, "learning_rate": 9.340289419824107e-05, "loss": 0.01623804122209549, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01637, "step": 111, "tokens/total": 3359440, "tokens/train_per_sec_per_gpu": 33.41, "tokens/trainable": 50708} +{"epoch": 0.4375, "grad_norm": 0.4614265263080597, "learning_rate": 9.325077903184159e-05, "loss": 0.021365080028772354, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.02159, "step": 112, "tokens/total": 3389472, "tokens/train_per_sec_per_gpu": 31.61, "tokens/trainable": 51154} +{"epoch": 0.44140625, "grad_norm": 0.28496578335762024, "learning_rate": 9.30970720943932e-05, "loss": 0.013247976079583168, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01334, "step": 113, "tokens/total": 3419920, "tokens/train_per_sec_per_gpu": 32.72, "tokens/trainable": 51610} +{"epoch": 0.4453125, "grad_norm": 0.5088275074958801, "learning_rate": 9.2941779782269e-05, "loss": 0.022269586101174355, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.02252, "step": 114, "tokens/total": 3450176, "tokens/train_per_sec_per_gpu": 29.14, "tokens/trainable": 52014} +{"epoch": 0.44921875, "grad_norm": 0.19142404198646545, "learning_rate": 9.278490855781596e-05, "loss": 0.006432589143514633, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00645, "step": 115, "tokens/total": 3480544, "tokens/train_per_sec_per_gpu": 36.32, "tokens/trainable": 52503} +{"epoch": 0.453125, "grad_norm": 1.382026195526123, "learning_rate": 9.262646494908604e-05, "loss": 0.012845169752836227, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01293, "step": 116, "tokens/total": 3510848, "tokens/train_per_sec_per_gpu": 38.75, "tokens/trainable": 53005} +{"epoch": 0.45703125, "grad_norm": 0.16703951358795166, "learning_rate": 9.246645554956457e-05, "loss": 0.0037160126958042383, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00372, "step": 117, "tokens/total": 3541296, "tokens/train_per_sec_per_gpu": 35.4, "tokens/trainable": 53464} +{"epoch": 0.4609375, "grad_norm": 0.615856945514679, "learning_rate": 9.230488701789578e-05, "loss": 0.014962448738515377, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.01507, "step": 118, "tokens/total": 3571936, "tokens/train_per_sec_per_gpu": 32.88, "tokens/trainable": 53920} +{"epoch": 0.46484375, "grad_norm": 0.24910427629947662, "learning_rate": 9.214176607760577e-05, "loss": 0.004819548688828945, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00483, "step": 119, "tokens/total": 3602000, "tokens/train_per_sec_per_gpu": 35.63, "tokens/trainable": 54360} +{"epoch": 0.46875, "grad_norm": 0.9213384389877319, "learning_rate": 9.197709951682268e-05, "loss": 0.008400149643421173, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00844, "step": 120, "tokens/total": 3632288, "tokens/train_per_sec_per_gpu": 35.21, "tokens/trainable": 54836} +{"epoch": 0.47265625, "grad_norm": 0.1228351816534996, "learning_rate": 9.181089418799428e-05, "loss": 0.0019936836324632168, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.002, "step": 121, "tokens/total": 3662608, "tokens/train_per_sec_per_gpu": 32.69, "tokens/trainable": 55264} +{"epoch": 0.4765625, "grad_norm": 0.18029853701591492, "learning_rate": 9.164315700760271e-05, "loss": 0.004047749564051628, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00406, "step": 122, "tokens/total": 3692864, "tokens/train_per_sec_per_gpu": 30.55, "tokens/trainable": 55697} +{"epoch": 0.48046875, "grad_norm": 0.4368670582771301, "learning_rate": 9.147389495587671e-05, "loss": 0.00867566466331482, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00871, "step": 123, "tokens/total": 3722864, "tokens/train_per_sec_per_gpu": 33.49, "tokens/trainable": 56141} +{"epoch": 0.484375, "grad_norm": 0.3315965235233307, "learning_rate": 9.130311507650116e-05, "loss": 0.011374368332326412, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.01144, "step": 124, "tokens/total": 3753056, "tokens/train_per_sec_per_gpu": 33.2, "tokens/trainable": 56586} +{"epoch": 0.48828125, "grad_norm": 0.2834460735321045, "learning_rate": 9.113082447632394e-05, "loss": 0.005234894342720509, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00525, "step": 125, "tokens/total": 3783312, "tokens/train_per_sec_per_gpu": 36.68, "tokens/trainable": 57078} +{"epoch": 0.4921875, "grad_norm": 0.5132001638412476, "learning_rate": 9.09570303250602e-05, "loss": 0.02469645068049431, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.025, "step": 126, "tokens/total": 3813712, "tokens/train_per_sec_per_gpu": 32.87, "tokens/trainable": 57524} +{"epoch": 0.49609375, "grad_norm": 0.9969733953475952, "learning_rate": 9.078173985499394e-05, "loss": 0.040580250322818756, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.04141, "step": 127, "tokens/total": 3844336, "tokens/train_per_sec_per_gpu": 34.85, "tokens/trainable": 58002} +{"epoch": 0.5, "grad_norm": 0.4804457724094391, "learning_rate": 9.060496036067713e-05, "loss": 0.007047833874821663, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.00707, "step": 128, "tokens/total": 3872688, "tokens/train_per_sec_per_gpu": 35.3, "tokens/trainable": 58451} +{"epoch": 0.50390625, "grad_norm": 0.23698025941848755, "learning_rate": 9.042669919862615e-05, "loss": 0.00366212404333055, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.64, "memory/max_allocated (GiB)": 33.64, "ppl": 1.00367, "step": 129, "tokens/total": 3902640, "tokens/train_per_sec_per_gpu": 32.1, "tokens/trainable": 58893} +{"epoch": 0.5078125, "grad_norm": 0.44587212800979614, "learning_rate": 9.024696378701557e-05, "loss": 0.010764295235276222, "memory/device_reserved (GiB)": 34.68, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01082, "step": 130, "tokens/total": 3933104, "tokens/train_per_sec_per_gpu": 35.94, "tokens/trainable": 59353} +{"epoch": 0.51171875, "grad_norm": 0.3212004005908966, "learning_rate": 9.006576160536948e-05, "loss": 0.013213565573096275, "memory/device_reserved (GiB)": 34.92, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.0133, "step": 131, "tokens/total": 3963184, "tokens/train_per_sec_per_gpu": 32.29, "tokens/trainable": 59789} +{"epoch": 0.515625, "grad_norm": 0.14710628986358643, "learning_rate": 8.988310019425035e-05, "loss": 0.005224708467721939, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00524, "step": 132, "tokens/total": 3993648, "tokens/train_per_sec_per_gpu": 35.82, "tokens/trainable": 60263} +{"epoch": 0.51953125, "grad_norm": 0.5616672039031982, "learning_rate": 8.969898715494506e-05, "loss": 0.01082855835556984, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01089, "step": 133, "tokens/total": 4024000, "tokens/train_per_sec_per_gpu": 35.55, "tokens/trainable": 60756} +{"epoch": 0.5234375, "grad_norm": 0.28706833720207214, "learning_rate": 8.951343014914869e-05, "loss": 0.02170894667506218, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.02195, "step": 134, "tokens/total": 4052480, "tokens/train_per_sec_per_gpu": 31.85, "tokens/trainable": 61201} +{"epoch": 0.52734375, "grad_norm": 0.1448316127061844, "learning_rate": 8.932643689864568e-05, "loss": 0.0064529795199632645, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00647, "step": 135, "tokens/total": 4082864, "tokens/train_per_sec_per_gpu": 36.07, "tokens/trainable": 61681} +{"epoch": 0.53125, "grad_norm": 0.2812711000442505, "learning_rate": 8.913801518498845e-05, "loss": 0.01243551354855299, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.7, "memory/max_allocated (GiB)": 33.7, "ppl": 1.01251, "step": 136, "tokens/total": 4113040, "tokens/train_per_sec_per_gpu": 38.33, "tokens/trainable": 62156} +{"epoch": 0.53515625, "grad_norm": 0.37281039357185364, "learning_rate": 8.894817284917364e-05, "loss": 0.009842045605182648, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00989, "step": 137, "tokens/total": 4143344, "tokens/train_per_sec_per_gpu": 35.42, "tokens/trainable": 62646} +{"epoch": 0.5390625, "grad_norm": 0.15259796380996704, "learning_rate": 8.875691779131569e-05, "loss": 0.007893089205026627, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00792, "step": 138, "tokens/total": 4173648, "tokens/train_per_sec_per_gpu": 36.58, "tokens/trainable": 63097} +{"epoch": 0.54296875, "grad_norm": 0.2609741985797882, "learning_rate": 8.856425797031829e-05, "loss": 0.009887185879051685, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00994, "step": 139, "tokens/total": 4204080, "tokens/train_per_sec_per_gpu": 37.62, "tokens/trainable": 63585} +{"epoch": 0.546875, "grad_norm": 0.3938979506492615, "learning_rate": 8.837020140354295e-05, "loss": 0.020112136378884315, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.42, "memory/max_allocated (GiB)": 33.42, "ppl": 1.02032, "step": 140, "tokens/total": 4232400, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 64002} +{"epoch": 0.55078125, "grad_norm": 0.2883419096469879, "learning_rate": 8.817475616647554e-05, "loss": 0.020921282470226288, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.02114, "step": 141, "tokens/total": 4262624, "tokens/train_per_sec_per_gpu": 35.28, "tokens/trainable": 64459} +{"epoch": 0.5546875, "grad_norm": 0.22204594314098358, "learning_rate": 8.797793039239017e-05, "loss": 0.007596657145768404, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00763, "step": 142, "tokens/total": 4293168, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 64897} +{"epoch": 0.55859375, "grad_norm": 0.3463776409626007, "learning_rate": 8.777973227201069e-05, "loss": 0.008941063657402992, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00898, "step": 143, "tokens/total": 4323168, "tokens/train_per_sec_per_gpu": 37.81, "tokens/trainable": 65376} +{"epoch": 0.5625, "grad_norm": 0.07992643117904663, "learning_rate": 8.758017005316988e-05, "loss": 0.002066058572381735, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00207, "step": 144, "tokens/total": 4353264, "tokens/train_per_sec_per_gpu": 36.81, "tokens/trainable": 65842} +{"epoch": 0.56640625, "grad_norm": 0.4542894959449768, "learning_rate": 8.737925204046629e-05, "loss": 0.006295363884419203, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00632, "step": 145, "tokens/total": 4383504, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 66282} +{"epoch": 0.5703125, "grad_norm": 0.6165645122528076, "learning_rate": 8.717698659491851e-05, "loss": 0.021687351167201996, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.02192, "step": 146, "tokens/total": 4413776, "tokens/train_per_sec_per_gpu": 31.67, "tokens/trainable": 66733} +{"epoch": 0.57421875, "grad_norm": 0.44622310996055603, "learning_rate": 8.697338213361735e-05, "loss": 0.010542265139520168, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0106, "step": 147, "tokens/total": 4444032, "tokens/train_per_sec_per_gpu": 35.41, "tokens/trainable": 67147} +{"epoch": 0.578125, "grad_norm": 0.5972599387168884, "learning_rate": 8.676844712937552e-05, "loss": 0.015109038911759853, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01522, "step": 148, "tokens/total": 4474384, "tokens/train_per_sec_per_gpu": 35.19, "tokens/trainable": 67630} +{"epoch": 0.58203125, "grad_norm": 0.16330066323280334, "learning_rate": 8.656219011037509e-05, "loss": 0.003038810333237052, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00304, "step": 149, "tokens/total": 4504800, "tokens/train_per_sec_per_gpu": 33.97, "tokens/trainable": 68074} +{"epoch": 0.5859375, "grad_norm": 0.1966242492198944, "learning_rate": 8.63546196598125e-05, "loss": 0.00474123377352953, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00475, "step": 150, "tokens/total": 4534896, "tokens/train_per_sec_per_gpu": 33.28, "tokens/trainable": 68542} +{"epoch": 0.58984375, "grad_norm": 0.2519090175628662, "learning_rate": 8.614574441554145e-05, "loss": 0.005299043841660023, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00531, "step": 151, "tokens/total": 4565008, "tokens/train_per_sec_per_gpu": 30.34, "tokens/trainable": 69008} +{"epoch": 0.59375, "grad_norm": 0.45198729634284973, "learning_rate": 8.593557306971349e-05, "loss": 0.00804843008518219, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00808, "step": 152, "tokens/total": 4595520, "tokens/train_per_sec_per_gpu": 35.52, "tokens/trainable": 69462} +{"epoch": 0.59765625, "grad_norm": 1.9722840785980225, "learning_rate": 8.572411436841618e-05, "loss": 0.015857091173529625, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01598, "step": 153, "tokens/total": 4625952, "tokens/train_per_sec_per_gpu": 33.96, "tokens/trainable": 69925} +{"epoch": 0.6015625, "grad_norm": 0.1563708782196045, "learning_rate": 8.551137711130922e-05, "loss": 0.003014163114130497, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00302, "step": 154, "tokens/total": 4656272, "tokens/train_per_sec_per_gpu": 36.73, "tokens/trainable": 70392} +{"epoch": 0.60546875, "grad_norm": 0.3952409327030182, "learning_rate": 8.529737015125824e-05, "loss": 0.020546773448586464, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.02076, "step": 155, "tokens/total": 4684592, "tokens/train_per_sec_per_gpu": 35.63, "tokens/trainable": 70826} +{"epoch": 0.609375, "grad_norm": 0.29893431067466736, "learning_rate": 8.508210239396639e-05, "loss": 0.006672243122011423, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00669, "step": 156, "tokens/total": 4715136, "tokens/train_per_sec_per_gpu": 34.03, "tokens/trainable": 71284} +{"epoch": 0.61328125, "grad_norm": 0.12646262347698212, "learning_rate": 8.486558279760375e-05, "loss": 0.001321017974987626, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00132, "step": 157, "tokens/total": 4745296, "tokens/train_per_sec_per_gpu": 35.12, "tokens/trainable": 71746} +{"epoch": 0.6171875, "grad_norm": 0.3663196563720703, "learning_rate": 8.464782037243449e-05, "loss": 0.00867768656462431, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00872, "step": 158, "tokens/total": 4775696, "tokens/train_per_sec_per_gpu": 32.33, "tokens/trainable": 72182} +{"epoch": 0.62109375, "grad_norm": 0.32793474197387695, "learning_rate": 8.442882418044202e-05, "loss": 0.012621743604540825, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.0127, "step": 159, "tokens/total": 4805776, "tokens/train_per_sec_per_gpu": 30.69, "tokens/trainable": 72601} +{"epoch": 0.625, "grad_norm": 0.502180278301239, "learning_rate": 8.420860333495179e-05, "loss": 0.02071015164256096, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.02093, "step": 160, "tokens/total": 4836224, "tokens/train_per_sec_per_gpu": 31.57, "tokens/trainable": 73058} +{"epoch": 0.62890625, "grad_norm": 3.3360722064971924, "learning_rate": 8.398716700025208e-05, "loss": 0.010355101898312569, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.01041, "step": 161, "tokens/total": 4866464, "tokens/train_per_sec_per_gpu": 30.11, "tokens/trainable": 73471} +{"epoch": 0.6328125, "grad_norm": 0.42214301228523254, "learning_rate": 8.376452439121266e-05, "loss": 0.010006662458181381, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.01006, "step": 162, "tokens/total": 4897040, "tokens/train_per_sec_per_gpu": 32.77, "tokens/trainable": 73920} +{"epoch": 0.63671875, "grad_norm": 0.5886772871017456, "learning_rate": 8.354068477290124e-05, "loss": 0.015542788431048393, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01566, "step": 163, "tokens/total": 4927360, "tokens/train_per_sec_per_gpu": 35.47, "tokens/trainable": 74376} +{"epoch": 0.640625, "grad_norm": 0.5640157461166382, "learning_rate": 8.331565746019807e-05, "loss": 0.019208863377571106, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01939, "step": 164, "tokens/total": 4957952, "tokens/train_per_sec_per_gpu": 33.45, "tokens/trainable": 74827} +{"epoch": 0.64453125, "grad_norm": 0.09329716116189957, "learning_rate": 8.308945181740812e-05, "loss": 0.001608099672012031, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00161, "step": 165, "tokens/total": 4988288, "tokens/train_per_sec_per_gpu": 31.79, "tokens/trainable": 75259} +{"epoch": 0.6484375, "grad_norm": 0.4663255512714386, "learning_rate": 8.286207725787153e-05, "loss": 0.003368590958416462, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00337, "step": 166, "tokens/total": 5018592, "tokens/train_per_sec_per_gpu": 31.92, "tokens/trainable": 75700} +{"epoch": 0.65234375, "grad_norm": 0.09389737993478775, "learning_rate": 8.263354324357182e-05, "loss": 0.004297872073948383, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00431, "step": 167, "tokens/total": 5049264, "tokens/train_per_sec_per_gpu": 31.38, "tokens/trainable": 76161} +{"epoch": 0.65625, "grad_norm": 0.2795652151107788, "learning_rate": 8.240385928474219e-05, "loss": 0.009957612492144108, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.01001, "step": 168, "tokens/total": 5079776, "tokens/train_per_sec_per_gpu": 34.52, "tokens/trainable": 76620} +{"epoch": 0.66015625, "grad_norm": 0.6108075976371765, "learning_rate": 8.217303493946967e-05, "loss": 0.004623654298484325, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00463, "step": 169, "tokens/total": 5110112, "tokens/train_per_sec_per_gpu": 34.47, "tokens/trainable": 77044} +{"epoch": 0.6640625, "grad_norm": 2.5686914920806885, "learning_rate": 8.194107981329746e-05, "loss": 0.014518939889967442, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01462, "step": 170, "tokens/total": 5140512, "tokens/train_per_sec_per_gpu": 34.2, "tokens/trainable": 77493} +{"epoch": 0.66796875, "grad_norm": 0.29714030027389526, "learning_rate": 8.170800355882518e-05, "loss": 0.010281096212565899, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.01033, "step": 171, "tokens/total": 5170768, "tokens/train_per_sec_per_gpu": 33.28, "tokens/trainable": 77953} +{"epoch": 0.671875, "grad_norm": 0.24662797152996063, "learning_rate": 8.147381587530713e-05, "loss": 0.005124978721141815, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00514, "step": 172, "tokens/total": 5201200, "tokens/train_per_sec_per_gpu": 29.9, "tokens/trainable": 78407} +{"epoch": 0.67578125, "grad_norm": 0.19063295423984528, "learning_rate": 8.123852650824877e-05, "loss": 0.005921770352870226, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00594, "step": 173, "tokens/total": 5231632, "tokens/train_per_sec_per_gpu": 33.16, "tokens/trainable": 78846} +{"epoch": 0.6796875, "grad_norm": 0.5131625533103943, "learning_rate": 8.100214524900103e-05, "loss": 0.01540191750973463, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.62, "memory/max_allocated (GiB)": 33.62, "ppl": 1.01552, "step": 174, "tokens/total": 5261360, "tokens/train_per_sec_per_gpu": 30.17, "tokens/trainable": 79263} +{"epoch": 0.68359375, "grad_norm": 0.11886871606111526, "learning_rate": 8.076468193435301e-05, "loss": 0.0013660730328410864, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00137, "step": 175, "tokens/total": 5291648, "tokens/train_per_sec_per_gpu": 32.59, "tokens/trainable": 79733} +{"epoch": 0.6875, "grad_norm": 0.39346420764923096, "learning_rate": 8.052614644612253e-05, "loss": 0.007847018539905548, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00788, "step": 176, "tokens/total": 5322160, "tokens/train_per_sec_per_gpu": 36.19, "tokens/trainable": 80211} +{"epoch": 0.69140625, "grad_norm": 0.05455538257956505, "learning_rate": 8.028654871074489e-05, "loss": 0.0007541990489698946, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00075, "step": 177, "tokens/total": 5352496, "tokens/train_per_sec_per_gpu": 32.69, "tokens/trainable": 80658} +{"epoch": 0.6953125, "grad_norm": 0.3038773536682129, "learning_rate": 8.004589869885986e-05, "loss": 0.006179198157042265, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.0062, "step": 178, "tokens/total": 5382704, "tokens/train_per_sec_per_gpu": 34.89, "tokens/trainable": 81126} +{"epoch": 0.69921875, "grad_norm": 0.08510788530111313, "learning_rate": 7.980420642489674e-05, "loss": 0.0012173540890216827, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00122, "step": 179, "tokens/total": 5412704, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 81573} +{"epoch": 0.703125, "grad_norm": 0.3349142074584961, "learning_rate": 7.95614819466576e-05, "loss": 0.00507398834452033, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00509, "step": 180, "tokens/total": 5442880, "tokens/train_per_sec_per_gpu": 34.64, "tokens/trainable": 82046} +{"epoch": 0.70703125, "grad_norm": 0.3200117349624634, "learning_rate": 7.931773536489872e-05, "loss": 0.007542713545262814, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00757, "step": 181, "tokens/total": 5473136, "tokens/train_per_sec_per_gpu": 38.92, "tokens/trainable": 82563} +{"epoch": 0.7109375, "grad_norm": 0.13203248381614685, "learning_rate": 7.907297682291035e-05, "loss": 0.0025186133570969105, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00252, "step": 182, "tokens/total": 5503488, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 83051} +{"epoch": 0.71484375, "grad_norm": 0.3643631637096405, "learning_rate": 7.882721650609442e-05, "loss": 0.012744799256324768, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01283, "step": 183, "tokens/total": 5533680, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 83498} +{"epoch": 0.71875, "grad_norm": 0.012706859968602657, "learning_rate": 7.85804646415409e-05, "loss": 0.00015147411613725126, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00015, "step": 184, "tokens/total": 5564160, "tokens/train_per_sec_per_gpu": 35.57, "tokens/trainable": 84004} +{"epoch": 0.72265625, "grad_norm": 0.14483962953090668, "learning_rate": 7.833273149760207e-05, "loss": 0.0014616945991292596, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00146, "step": 185, "tokens/total": 5594368, "tokens/train_per_sec_per_gpu": 35.63, "tokens/trainable": 84488} +{"epoch": 0.7265625, "grad_norm": 0.3152420222759247, "learning_rate": 7.808402738346527e-05, "loss": 0.005956828128546476, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00597, "step": 186, "tokens/total": 5624736, "tokens/train_per_sec_per_gpu": 33.45, "tokens/trainable": 84949} +{"epoch": 0.73046875, "grad_norm": 0.15444505214691162, "learning_rate": 7.783436264872382e-05, "loss": 0.0022030179388821125, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00221, "step": 187, "tokens/total": 5655024, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 85397} +{"epoch": 0.734375, "grad_norm": 1.1173254251480103, "learning_rate": 7.758374768294647e-05, "loss": 0.018108580261468887, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01827, "step": 188, "tokens/total": 5685280, "tokens/train_per_sec_per_gpu": 34.04, "tokens/trainable": 85886} +{"epoch": 0.73828125, "grad_norm": 0.15292231738567352, "learning_rate": 7.733219291524489e-05, "loss": 0.0033938095439225435, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.0034, "step": 189, "tokens/total": 5715568, "tokens/train_per_sec_per_gpu": 36.24, "tokens/trainable": 86393} +{"epoch": 0.7421875, "grad_norm": 0.3460272550582886, "learning_rate": 7.707970881383977e-05, "loss": 0.003943222109228373, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00395, "step": 190, "tokens/total": 5746160, "tokens/train_per_sec_per_gpu": 33.86, "tokens/trainable": 86864} +{"epoch": 0.74609375, "grad_norm": 0.17786552011966705, "learning_rate": 7.682630588562518e-05, "loss": 0.002941778162494302, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00295, "step": 191, "tokens/total": 5776736, "tokens/train_per_sec_per_gpu": 36.08, "tokens/trainable": 87312} +{"epoch": 0.75, "grad_norm": 0.05713065341114998, "learning_rate": 7.657199467573129e-05, "loss": 0.0006791478954255581, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00068, "step": 192, "tokens/total": 5807168, "tokens/train_per_sec_per_gpu": 30.28, "tokens/trainable": 87734} +{"epoch": 0.75390625, "grad_norm": 0.21285323798656464, "learning_rate": 7.631678576708561e-05, "loss": 0.005839701741933823, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00586, "step": 193, "tokens/total": 5837600, "tokens/train_per_sec_per_gpu": 35.12, "tokens/trainable": 88217} +{"epoch": 0.7578125, "grad_norm": 0.046665649861097336, "learning_rate": 7.606068977997255e-05, "loss": 0.0009576653246767819, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00096, "step": 194, "tokens/total": 5867920, "tokens/train_per_sec_per_gpu": 36.11, "tokens/trainable": 88697} +{"epoch": 0.76171875, "grad_norm": 0.12541399896144867, "learning_rate": 7.580371737159148e-05, "loss": 0.002426251769065857, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.38, "memory/max_allocated (GiB)": 33.38, "ppl": 1.00243, "step": 195, "tokens/total": 5896144, "tokens/train_per_sec_per_gpu": 30.41, "tokens/trainable": 89130} +{"epoch": 0.765625, "grad_norm": 0.06126544624567032, "learning_rate": 7.554587923561324e-05, "loss": 0.000532104168087244, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00053, "step": 196, "tokens/total": 5926544, "tokens/train_per_sec_per_gpu": 37.18, "tokens/trainable": 89592} +{"epoch": 0.76953125, "grad_norm": 0.11023399233818054, "learning_rate": 7.528718610173511e-05, "loss": 0.0028196191415190697, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.34, "memory/max_allocated (GiB)": 33.34, "ppl": 1.00282, "step": 197, "tokens/total": 5954864, "tokens/train_per_sec_per_gpu": 38.04, "tokens/trainable": 90040} +{"epoch": 0.7734375, "grad_norm": 0.16828177869319916, "learning_rate": 7.502764873523431e-05, "loss": 0.004482921212911606, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00449, "step": 198, "tokens/total": 5985120, "tokens/train_per_sec_per_gpu": 34.21, "tokens/trainable": 90501} +{"epoch": 0.77734375, "grad_norm": 0.19191870093345642, "learning_rate": 7.476727793652011e-05, "loss": 0.002102922648191452, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 34.0, "memory/max_allocated (GiB)": 34.0, "ppl": 1.00211, "step": 199, "tokens/total": 6015872, "tokens/train_per_sec_per_gpu": 36.51, "tokens/trainable": 90980} +{"epoch": 0.78125, "grad_norm": 0.0961354449391365, "learning_rate": 7.450608454068415e-05, "loss": 0.0013840183382853866, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00138, "step": 200, "tokens/total": 6046128, "tokens/train_per_sec_per_gpu": 39.36, "tokens/trainable": 91464} +{"epoch": 0.78515625, "grad_norm": 0.03724909946322441, "learning_rate": 7.424407941704987e-05, "loss": 0.00039239737088792026, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00039, "step": 201, "tokens/total": 6076128, "tokens/train_per_sec_per_gpu": 34.21, "tokens/trainable": 91910} +{"epoch": 0.7890625, "grad_norm": 0.47981590032577515, "learning_rate": 7.398127346871986e-05, "loss": 0.008898628875613213, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00894, "step": 202, "tokens/total": 6106480, "tokens/train_per_sec_per_gpu": 37.79, "tokens/trainable": 92383} +{"epoch": 0.79296875, "grad_norm": 0.25382256507873535, "learning_rate": 7.371767763212238e-05, "loss": 0.02967994660139084, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.03012, "step": 203, "tokens/total": 6136560, "tokens/train_per_sec_per_gpu": 36.6, "tokens/trainable": 92840} +{"epoch": 0.796875, "grad_norm": 0.11337354779243469, "learning_rate": 7.345330287655617e-05, "loss": 0.0005227055517025292, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00052, "step": 204, "tokens/total": 6167104, "tokens/train_per_sec_per_gpu": 34.02, "tokens/trainable": 93321} +{"epoch": 0.80078125, "grad_norm": 0.08194643259048462, "learning_rate": 7.31881602037339e-05, "loss": 0.0015989269595593214, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.0016, "step": 205, "tokens/total": 6197296, "tokens/train_per_sec_per_gpu": 34.8, "tokens/trainable": 93754} +{"epoch": 0.8046875, "grad_norm": 0.42092061042785645, "learning_rate": 7.29222606473245e-05, "loss": 0.005056938622146845, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00507, "step": 206, "tokens/total": 6227520, "tokens/train_per_sec_per_gpu": 37.51, "tokens/trainable": 94233} +{"epoch": 0.80859375, "grad_norm": 0.28918036818504333, "learning_rate": 7.265561527249383e-05, "loss": 0.00922542903572321, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00927, "step": 207, "tokens/total": 6257680, "tokens/train_per_sec_per_gpu": 36.57, "tokens/trainable": 94731} +{"epoch": 0.8125, "grad_norm": 0.052897270768880844, "learning_rate": 7.238823517544436e-05, "loss": 0.0008248636149801314, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00083, "step": 208, "tokens/total": 6287968, "tokens/train_per_sec_per_gpu": 33.76, "tokens/trainable": 95216} +{"epoch": 0.81640625, "grad_norm": 0.053124021738767624, "learning_rate": 7.212013148295333e-05, "loss": 0.0011008362052962184, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.0011, "step": 209, "tokens/total": 6318608, "tokens/train_per_sec_per_gpu": 29.28, "tokens/trainable": 95617} +{"epoch": 0.8203125, "grad_norm": 0.3910926580429077, "learning_rate": 7.185131535190975e-05, "loss": 0.0025537661276757717, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00256, "step": 210, "tokens/total": 6348864, "tokens/train_per_sec_per_gpu": 35.44, "tokens/trainable": 96074} +{"epoch": 0.82421875, "grad_norm": 0.1602534055709839, "learning_rate": 7.158179796885005e-05, "loss": 0.0042940047569572926, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.0043, "step": 211, "tokens/total": 6378880, "tokens/train_per_sec_per_gpu": 35.4, "tokens/trainable": 96503} +{"epoch": 0.828125, "grad_norm": 0.22192947566509247, "learning_rate": 7.131159054949273e-05, "loss": 0.002052969066426158, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00206, "step": 212, "tokens/total": 6409392, "tokens/train_per_sec_per_gpu": 32.73, "tokens/trainable": 96928} +{"epoch": 0.83203125, "grad_norm": 0.17284627258777618, "learning_rate": 7.104070433827139e-05, "loss": 0.0034506958909332752, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00346, "step": 213, "tokens/total": 6439824, "tokens/train_per_sec_per_gpu": 32.77, "tokens/trainable": 97359} +{"epoch": 0.8359375, "grad_norm": 0.2631441056728363, "learning_rate": 7.076915060786705e-05, "loss": 0.0022546069230884314, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00226, "step": 214, "tokens/total": 6470096, "tokens/train_per_sec_per_gpu": 36.38, "tokens/trainable": 97856} +{"epoch": 0.83984375, "grad_norm": 0.24480876326560974, "learning_rate": 7.049694065873882e-05, "loss": 0.005642582196742296, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00566, "step": 215, "tokens/total": 6500336, "tokens/train_per_sec_per_gpu": 37.45, "tokens/trainable": 98306} +{"epoch": 0.84375, "grad_norm": 0.01338098756968975, "learning_rate": 7.022408581865382e-05, "loss": 0.0002527001779526472, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00025, "step": 216, "tokens/total": 6530688, "tokens/train_per_sec_per_gpu": 36.48, "tokens/trainable": 98779} +{"epoch": 0.84765625, "grad_norm": 0.01576339825987816, "learning_rate": 6.99505974422157e-05, "loss": 0.00029742170590907335, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.0003, "step": 217, "tokens/total": 6561040, "tokens/train_per_sec_per_gpu": 36.78, "tokens/trainable": 99264} +{"epoch": 0.8515625, "grad_norm": 0.14224660396575928, "learning_rate": 6.967648691039213e-05, "loss": 0.0019305831519886851, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00193, "step": 218, "tokens/total": 6589392, "tokens/train_per_sec_per_gpu": 39.69, "tokens/trainable": 99729} +{"epoch": 0.85546875, "grad_norm": 0.13956739008426666, "learning_rate": 6.940176563004123e-05, "loss": 0.0020216472912579775, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00202, "step": 219, "tokens/total": 6619824, "tokens/train_per_sec_per_gpu": 31.73, "tokens/trainable": 100162} +{"epoch": 0.859375, "grad_norm": 0.2409037947654724, "learning_rate": 6.912644503343682e-05, "loss": 0.0016943392110988498, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.0017, "step": 220, "tokens/total": 6650224, "tokens/train_per_sec_per_gpu": 36.42, "tokens/trainable": 100619} +{"epoch": 0.86328125, "grad_norm": 0.004593817982822657, "learning_rate": 6.885053657779273e-05, "loss": 0.00010279623529640958, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0001, "step": 221, "tokens/total": 6680384, "tokens/train_per_sec_per_gpu": 32.93, "tokens/trainable": 101079} +{"epoch": 0.8671875, "grad_norm": 0.3027840256690979, "learning_rate": 6.857405174478604e-05, "loss": 0.0038326543290168047, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00384, "step": 222, "tokens/total": 6710896, "tokens/train_per_sec_per_gpu": 37.86, "tokens/trainable": 101545} +{"epoch": 0.87109375, "grad_norm": 0.06953064352273941, "learning_rate": 6.82970020400792e-05, "loss": 0.0005575605318881571, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00056, "step": 223, "tokens/total": 6741280, "tokens/train_per_sec_per_gpu": 31.6, "tokens/trainable": 101986} +{"epoch": 0.875, "grad_norm": 0.14961817860603333, "learning_rate": 6.801939899284132e-05, "loss": 0.0019154315814375877, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00192, "step": 224, "tokens/total": 6771488, "tokens/train_per_sec_per_gpu": 35.51, "tokens/trainable": 102434} +{"epoch": 0.87890625, "grad_norm": 0.1315905600786209, "learning_rate": 6.774125415526827e-05, "loss": 0.0011086631566286087, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00111, "step": 225, "tokens/total": 6801744, "tokens/train_per_sec_per_gpu": 29.24, "tokens/trainable": 102842} +{"epoch": 0.8828125, "grad_norm": 0.7103981375694275, "learning_rate": 6.746257910210214e-05, "loss": 0.007341520860791206, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00737, "step": 226, "tokens/total": 6832432, "tokens/train_per_sec_per_gpu": 37.49, "tokens/trainable": 103341} +{"epoch": 0.88671875, "grad_norm": 0.014314249157905579, "learning_rate": 6.718338543014937e-05, "loss": 0.0001612441410543397, "memory/device_reserved (GiB)": 35.71, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00016, "step": 227, "tokens/total": 6862448, "tokens/train_per_sec_per_gpu": 40.93, "tokens/trainable": 103843} +{"epoch": 0.890625, "grad_norm": 0.00822363793849945, "learning_rate": 6.69036847577983e-05, "loss": 9.871406655292958e-05, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.0001, "step": 228, "tokens/total": 6892832, "tokens/train_per_sec_per_gpu": 36.35, "tokens/trainable": 104306} +{"epoch": 0.89453125, "grad_norm": 0.08351919800043106, "learning_rate": 6.662348872453553e-05, "loss": 0.0004943685489706695, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00049, "step": 229, "tokens/total": 6922832, "tokens/train_per_sec_per_gpu": 33.47, "tokens/trainable": 104771} +{"epoch": 0.8984375, "grad_norm": 0.004186064004898071, "learning_rate": 6.63428089904618e-05, "loss": 6.949146336410195e-05, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00007, "step": 230, "tokens/total": 6953104, "tokens/train_per_sec_per_gpu": 36.91, "tokens/trainable": 105239} +{"epoch": 0.90234375, "grad_norm": 0.008335944265127182, "learning_rate": 6.60616572358065e-05, "loss": 0.00010860025940928608, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00011, "step": 231, "tokens/total": 6983440, "tokens/train_per_sec_per_gpu": 33.87, "tokens/trainable": 105702} +{"epoch": 0.90625, "grad_norm": 0.11052291095256805, "learning_rate": 6.578004516044172e-05, "loss": 0.0004777174908667803, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00048, "step": 232, "tokens/total": 7013840, "tokens/train_per_sec_per_gpu": 33.12, "tokens/trainable": 106178} +{"epoch": 0.91015625, "grad_norm": 0.28182464838027954, "learning_rate": 6.549798448339548e-05, "loss": 0.0048217386938631535, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00483, "step": 233, "tokens/total": 7044240, "tokens/train_per_sec_per_gpu": 38.1, "tokens/trainable": 106682} +{"epoch": 0.9140625, "grad_norm": 0.19313767552375793, "learning_rate": 6.521548694236384e-05, "loss": 0.0006764436257071793, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00068, "step": 234, "tokens/total": 7074224, "tokens/train_per_sec_per_gpu": 33.95, "tokens/trainable": 107120} +{"epoch": 0.91796875, "grad_norm": 0.005185638088732958, "learning_rate": 6.493256429322259e-05, "loss": 6.307615694822744e-05, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00006, "step": 235, "tokens/total": 7104672, "tokens/train_per_sec_per_gpu": 32.8, "tokens/trainable": 107560} +{"epoch": 0.921875, "grad_norm": 0.07209689170122147, "learning_rate": 6.464922830953799e-05, "loss": 0.0004264797898940742, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00043, "step": 236, "tokens/total": 7134784, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 108017} +{"epoch": 0.92578125, "grad_norm": 0.005625654011964798, "learning_rate": 6.436549078207688e-05, "loss": 4.9979436880676076e-05, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00005, "step": 237, "tokens/total": 7164960, "tokens/train_per_sec_per_gpu": 33.47, "tokens/trainable": 108445} +{"epoch": 0.9296875, "grad_norm": 0.008250541053712368, "learning_rate": 6.408136351831592e-05, "loss": 6.751110777258873e-05, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00007, "step": 238, "tokens/total": 7195344, "tokens/train_per_sec_per_gpu": 29.25, "tokens/trainable": 108865} +{"epoch": 0.93359375, "grad_norm": 0.5217323899269104, "learning_rate": 6.379685834195036e-05, "loss": 0.0005912262131460011, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00059, "step": 239, "tokens/total": 7225680, "tokens/train_per_sec_per_gpu": 33.97, "tokens/trainable": 109315} +{"epoch": 0.9375, "grad_norm": 0.011697136797010899, "learning_rate": 6.351198709240186e-05, "loss": 9.851530194282532e-05, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0001, "step": 240, "tokens/total": 7256112, "tokens/train_per_sec_per_gpu": 33.41, "tokens/trainable": 109786} +{"epoch": 0.94140625, "grad_norm": 0.008918526582419872, "learning_rate": 6.32267616243259e-05, "loss": 6.70133886160329e-05, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00007, "step": 241, "tokens/total": 7286384, "tokens/train_per_sec_per_gpu": 38.51, "tokens/trainable": 110245} +{"epoch": 0.9453125, "grad_norm": 1.0020146369934082, "learning_rate": 6.294119380711849e-05, "loss": 0.013925625942647457, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01402, "step": 242, "tokens/total": 7316688, "tokens/train_per_sec_per_gpu": 29.29, "tokens/trainable": 110666} +{"epoch": 0.94921875, "grad_norm": 0.00132578460033983, "learning_rate": 6.265529552442209e-05, "loss": 1.891679858090356e-05, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00002, "step": 243, "tokens/total": 7346992, "tokens/train_per_sec_per_gpu": 38.21, "tokens/trainable": 111153} +{"epoch": 0.953125, "grad_norm": 0.898801326751709, "learning_rate": 6.236907867363127e-05, "loss": 0.012029297649860382, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0121, "step": 244, "tokens/total": 7377632, "tokens/train_per_sec_per_gpu": 33.15, "tokens/trainable": 111610} +{"epoch": 0.95703125, "grad_norm": 0.17281004786491394, "learning_rate": 6.208255516539749e-05, "loss": 0.001876274705864489, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00188, "step": 245, "tokens/total": 7407872, "tokens/train_per_sec_per_gpu": 34.28, "tokens/trainable": 112101} +{"epoch": 0.9609375, "grad_norm": 0.8393605947494507, "learning_rate": 6.179573692313344e-05, "loss": 0.02536478079855442, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.02569, "step": 246, "tokens/total": 7438448, "tokens/train_per_sec_per_gpu": 33.3, "tokens/trainable": 112568} +{"epoch": 0.96484375, "grad_norm": 0.1928865760564804, "learning_rate": 6.150863588251694e-05, "loss": 0.001368399360217154, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00137, "step": 247, "tokens/total": 7466768, "tokens/train_per_sec_per_gpu": 37.8, "tokens/trainable": 113022} +{"epoch": 0.96875, "grad_norm": 0.11117105931043625, "learning_rate": 6.122126399099419e-05, "loss": 0.0006689627189189196, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00067, "step": 248, "tokens/total": 7497360, "tokens/train_per_sec_per_gpu": 35.32, "tokens/trainable": 113463} +{"epoch": 0.97265625, "grad_norm": 0.041484661400318146, "learning_rate": 6.0933633207282615e-05, "loss": 0.0006179807242006063, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00062, "step": 249, "tokens/total": 7527504, "tokens/train_per_sec_per_gpu": 32.69, "tokens/trainable": 113907} +{"epoch": 0.9765625, "grad_norm": 0.0462549552321434, "learning_rate": 6.064575550087316e-05, "loss": 0.000625512795522809, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00063, "step": 250, "tokens/total": 7555808, "tokens/train_per_sec_per_gpu": 35.42, "tokens/trainable": 114347} +{"epoch": 0.98046875, "grad_norm": 0.23392240703105927, "learning_rate": 6.0357642851532245e-05, "loss": 0.004232748411595821, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00424, "step": 251, "tokens/total": 7586320, "tokens/train_per_sec_per_gpu": 35.32, "tokens/trainable": 114829} +{"epoch": 0.984375, "grad_norm": 0.3223143517971039, "learning_rate": 6.0069307248803294e-05, "loss": 0.007172638550400734, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0072, "step": 252, "tokens/total": 7616576, "tokens/train_per_sec_per_gpu": 35.27, "tokens/trainable": 115277} +{"epoch": 0.98828125, "grad_norm": 0.1359666883945465, "learning_rate": 5.9780760691507635e-05, "loss": 0.003720771986991167, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00373, "step": 253, "tokens/total": 7647152, "tokens/train_per_sec_per_gpu": 32.4, "tokens/trainable": 115698} +{"epoch": 0.9921875, "grad_norm": 0.10392674058675766, "learning_rate": 5.9492015187245334e-05, "loss": 0.0012953465338796377, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.0013, "step": 254, "tokens/total": 7677584, "tokens/train_per_sec_per_gpu": 30.89, "tokens/trainable": 116132} +{"epoch": 0.99609375, "grad_norm": 0.1488286703824997, "learning_rate": 5.920308275189541e-05, "loss": 0.0010544590186327696, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00106, "step": 255, "tokens/total": 7708016, "tokens/train_per_sec_per_gpu": 36.88, "tokens/trainable": 116590} +{"epoch": 1.0, "grad_norm": 0.025846293196082115, "learning_rate": 5.8913975409115874e-05, "loss": 0.00045689160469919443, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00046, "step": 256, "tokens/total": 7738608, "tokens/train_per_sec_per_gpu": 32.19, "tokens/trainable": 117038} +{"epoch": 1.00390625, "grad_norm": 0.011676807887852192, "learning_rate": 5.8624705189843395e-05, "loss": 0.00018332478066440672, "memory/device_reserved (GiB)": 36.13, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00018, "step": 257, "tokens/total": 7768944, "tokens/train_per_sec_per_gpu": 32.51, "tokens/trainable": 117494} +{"epoch": 1.0078125, "grad_norm": 0.026530586183071136, "learning_rate": 5.833528413179249e-05, "loss": 0.0004955548793077469, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0005, "step": 258, "tokens/total": 7799232, "tokens/train_per_sec_per_gpu": 30.63, "tokens/trainable": 117924} +{"epoch": 1.01171875, "grad_norm": 0.03306467458605766, "learning_rate": 5.80457242789548e-05, "loss": 0.0007393938140012324, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00074, "step": 259, "tokens/total": 7829568, "tokens/train_per_sec_per_gpu": 31.76, "tokens/trainable": 118372} +{"epoch": 1.015625, "grad_norm": 0.01490688230842352, "learning_rate": 5.77560376810977e-05, "loss": 0.00033902074210345745, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00034, "step": 260, "tokens/total": 7860000, "tokens/train_per_sec_per_gpu": 31.54, "tokens/trainable": 118815} +{"epoch": 1.01953125, "grad_norm": 0.23672910034656525, "learning_rate": 5.7466236393263005e-05, "loss": 0.003403924172744155, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00341, "step": 261, "tokens/total": 7890464, "tokens/train_per_sec_per_gpu": 33.82, "tokens/trainable": 119262} +{"epoch": 1.0234375, "grad_norm": 0.11774802953004837, "learning_rate": 5.717633247526522e-05, "loss": 0.002715344773605466, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00272, "step": 262, "tokens/total": 7920944, "tokens/train_per_sec_per_gpu": 34.86, "tokens/trainable": 119771} +{"epoch": 1.02734375, "grad_norm": 0.028089547529816628, "learning_rate": 5.688633799118971e-05, "loss": 0.0006553111597895622, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00066, "step": 263, "tokens/total": 7951296, "tokens/train_per_sec_per_gpu": 33.9, "tokens/trainable": 120268} +{"epoch": 1.03125, "grad_norm": 0.04741915315389633, "learning_rate": 5.659626500889066e-05, "loss": 0.0009464840404689312, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00095, "step": 264, "tokens/total": 7981856, "tokens/train_per_sec_per_gpu": 37.13, "tokens/trainable": 120745} +{"epoch": 1.03515625, "grad_norm": 0.06473700702190399, "learning_rate": 5.6306125599488905e-05, "loss": 0.0010454395087435842, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00105, "step": 265, "tokens/total": 8012192, "tokens/train_per_sec_per_gpu": 38.37, "tokens/trainable": 121203} +{"epoch": 1.0390625, "grad_norm": 0.1152043268084526, "learning_rate": 5.601593183686955e-05, "loss": 0.0010041436180472374, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.001, "step": 266, "tokens/total": 8042736, "tokens/train_per_sec_per_gpu": 31.29, "tokens/trainable": 121669} +{"epoch": 1.04296875, "grad_norm": 0.1012289822101593, "learning_rate": 5.572569579717961e-05, "loss": 0.0007158135995268822, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00072, "step": 267, "tokens/total": 8073024, "tokens/train_per_sec_per_gpu": 40.14, "tokens/trainable": 122182} +{"epoch": 1.046875, "grad_norm": 0.014074633829295635, "learning_rate": 5.543542955832538e-05, "loss": 0.00022851164976600558, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00023, "step": 268, "tokens/total": 8103280, "tokens/train_per_sec_per_gpu": 35.46, "tokens/trainable": 122644} +{"epoch": 1.05078125, "grad_norm": 0.022633297368884087, "learning_rate": 5.514514519946986e-05, "loss": 0.00056973792379722, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00057, "step": 269, "tokens/total": 8133440, "tokens/train_per_sec_per_gpu": 33.16, "tokens/trainable": 123104} +{"epoch": 1.0546875, "grad_norm": 0.028963575139641762, "learning_rate": 5.485485480053015e-05, "loss": 0.00030712541774846613, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00031, "step": 270, "tokens/total": 8163920, "tokens/train_per_sec_per_gpu": 36.27, "tokens/trainable": 123594} +{"epoch": 1.05859375, "grad_norm": 0.05075100436806679, "learning_rate": 5.4564570441674645e-05, "loss": 0.0004000376502517611, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.0004, "step": 271, "tokens/total": 8194032, "tokens/train_per_sec_per_gpu": 37.71, "tokens/trainable": 124076} +{"epoch": 1.0625, "grad_norm": 0.014762450009584427, "learning_rate": 5.42743042028204e-05, "loss": 0.0001975473714992404, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0002, "step": 272, "tokens/total": 8224320, "tokens/train_per_sec_per_gpu": 34.82, "tokens/trainable": 124546} +{"epoch": 1.06640625, "grad_norm": 0.3096662759780884, "learning_rate": 5.3984068163130464e-05, "loss": 0.0021046444308012724, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00211, "step": 273, "tokens/total": 8254368, "tokens/train_per_sec_per_gpu": 34.1, "tokens/trainable": 124976} +{"epoch": 1.0703125, "grad_norm": 0.14023423194885254, "learning_rate": 5.369387440051111e-05, "loss": 0.0011167863849550486, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00112, "step": 274, "tokens/total": 8284944, "tokens/train_per_sec_per_gpu": 34.67, "tokens/trainable": 125437} +{"epoch": 1.07421875, "grad_norm": 0.006373860873281956, "learning_rate": 5.340373499110935e-05, "loss": 0.00010782821482280269, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00011, "step": 275, "tokens/total": 8315584, "tokens/train_per_sec_per_gpu": 31.54, "tokens/trainable": 125888} +{"epoch": 1.078125, "grad_norm": 0.44675078988075256, "learning_rate": 5.3113662008810304e-05, "loss": 0.0035920855589210987, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0036, "step": 276, "tokens/total": 8345952, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 126362} +{"epoch": 1.08203125, "grad_norm": 0.0032541481778025627, "learning_rate": 5.282366752473479e-05, "loss": 5.1655006245709956e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00005, "step": 277, "tokens/total": 8376416, "tokens/train_per_sec_per_gpu": 34.01, "tokens/trainable": 126827} +{"epoch": 1.0859375, "grad_norm": 0.14338454604148865, "learning_rate": 5.2533763606737005e-05, "loss": 0.0026056424248963594, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00261, "step": 278, "tokens/total": 8406640, "tokens/train_per_sec_per_gpu": 38.13, "tokens/trainable": 127330} +{"epoch": 1.08984375, "grad_norm": 0.20478253066539764, "learning_rate": 5.224396231890232e-05, "loss": 0.0012996959267184138, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.0013, "step": 279, "tokens/total": 8436784, "tokens/train_per_sec_per_gpu": 34.44, "tokens/trainable": 127797} +{"epoch": 1.09375, "grad_norm": 0.0009973018895834684, "learning_rate": 5.195427572104522e-05, "loss": 2.559608401497826e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00003, "step": 280, "tokens/total": 8467152, "tokens/train_per_sec_per_gpu": 32.87, "tokens/trainable": 128246} +{"epoch": 1.09765625, "grad_norm": 0.001558113843202591, "learning_rate": 5.166471586820751e-05, "loss": 3.741440741578117e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00004, "step": 281, "tokens/total": 8497424, "tokens/train_per_sec_per_gpu": 36.25, "tokens/trainable": 128728} +{"epoch": 1.1015625, "grad_norm": 0.01588490605354309, "learning_rate": 5.1375294810156615e-05, "loss": 8.873045590007678e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00009, "step": 282, "tokens/total": 8527616, "tokens/train_per_sec_per_gpu": 35.14, "tokens/trainable": 129202} +{"epoch": 1.10546875, "grad_norm": 0.06619437038898468, "learning_rate": 5.1086024590884144e-05, "loss": 0.0010551117593422532, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00106, "step": 283, "tokens/total": 8557984, "tokens/train_per_sec_per_gpu": 33.09, "tokens/trainable": 129652} +{"epoch": 1.109375, "grad_norm": 0.1938449889421463, "learning_rate": 5.079691724810461e-05, "loss": 0.0008955710800364614, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.0009, "step": 284, "tokens/total": 8588384, "tokens/train_per_sec_per_gpu": 32.72, "tokens/trainable": 130099} +{"epoch": 1.11328125, "grad_norm": 0.16278043389320374, "learning_rate": 5.0507984812754684e-05, "loss": 0.0006343181594274938, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00063, "step": 285, "tokens/total": 8618768, "tokens/train_per_sec_per_gpu": 30.56, "tokens/trainable": 130542} +{"epoch": 1.1171875, "grad_norm": 0.0029784520156681538, "learning_rate": 5.021923930849237e-05, "loss": 4.526478369371034e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00005, "step": 286, "tokens/total": 8648976, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 130999} +{"epoch": 1.12109375, "grad_norm": 0.06447792798280716, "learning_rate": 4.99306927511967e-05, "loss": 5.6983706599567086e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00006, "step": 287, "tokens/total": 8679152, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 131448} +{"epoch": 1.125, "grad_norm": 0.00280313054099679, "learning_rate": 4.964235714846775e-05, "loss": 4.409470420796424e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00004, "step": 288, "tokens/total": 8709504, "tokens/train_per_sec_per_gpu": 36.04, "tokens/trainable": 131922} +{"epoch": 1.12890625, "grad_norm": 0.0028473951388150454, "learning_rate": 4.9354244499126866e-05, "loss": 3.198662307113409e-05, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00003, "step": 289, "tokens/total": 8739680, "tokens/train_per_sec_per_gpu": 33.26, "tokens/trainable": 132396} +{"epoch": 1.1328125, "grad_norm": 0.03604506328701973, "learning_rate": 4.90663667927174e-05, "loss": 0.00023753194545861334, "memory/device_reserved (GiB)": 35.57, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00024, "step": 290, "tokens/total": 8770112, "tokens/train_per_sec_per_gpu": 36.46, "tokens/trainable": 132845} +{"epoch": 1.13671875, "grad_norm": 0.2052794098854065, "learning_rate": 4.877873600900581e-05, "loss": 0.0011388716520741582, "memory/device_reserved (GiB)": 35.57, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00114, "step": 291, "tokens/total": 8800368, "tokens/train_per_sec_per_gpu": 35.81, "tokens/trainable": 133346} +{"epoch": 1.140625, "grad_norm": 0.12911155819892883, "learning_rate": 4.849136411748306e-05, "loss": 0.00046830251812934875, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00047, "step": 292, "tokens/total": 8830688, "tokens/train_per_sec_per_gpu": 34.38, "tokens/trainable": 133778} +{"epoch": 1.14453125, "grad_norm": 0.0032441529911011457, "learning_rate": 4.8204263076866574e-05, "loss": 4.348478250904009e-05, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00004, "step": 293, "tokens/total": 8861008, "tokens/train_per_sec_per_gpu": 36.39, "tokens/trainable": 134250} +{"epoch": 1.1484375, "grad_norm": 0.1916627436876297, "learning_rate": 4.791744483460251e-05, "loss": 0.0007836729055270553, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00078, "step": 294, "tokens/total": 8891760, "tokens/train_per_sec_per_gpu": 33.57, "tokens/trainable": 134682} +{"epoch": 1.15234375, "grad_norm": 0.11497358977794647, "learning_rate": 4.7630921326368736e-05, "loss": 0.0007789967930875719, "memory/device_reserved (GiB)": 35.59, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00078, "step": 295, "tokens/total": 8922192, "tokens/train_per_sec_per_gpu": 35.57, "tokens/trainable": 135164} +{"epoch": 1.15625, "grad_norm": 0.07835118472576141, "learning_rate": 4.7344704475577916e-05, "loss": 0.00046658708015456796, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00047, "step": 296, "tokens/total": 8952496, "tokens/train_per_sec_per_gpu": 33.28, "tokens/trainable": 135598} +{"epoch": 1.16015625, "grad_norm": 0.43025892972946167, "learning_rate": 4.705880619288153e-05, "loss": 0.011139214970171452, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.0112, "step": 297, "tokens/total": 8982768, "tokens/train_per_sec_per_gpu": 32.7, "tokens/trainable": 136037} +{"epoch": 1.1640625, "grad_norm": 0.0018146105576306581, "learning_rate": 4.677323837567412e-05, "loss": 2.296476304763928e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00002, "step": 298, "tokens/total": 9013248, "tokens/train_per_sec_per_gpu": 32.97, "tokens/trainable": 136494} +{"epoch": 1.16796875, "grad_norm": 0.013126488775014877, "learning_rate": 4.6488012907598146e-05, "loss": 9.889354987535626e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0001, "step": 299, "tokens/total": 9043632, "tokens/train_per_sec_per_gpu": 35.41, "tokens/trainable": 136952} +{"epoch": 1.171875, "grad_norm": 0.04675251618027687, "learning_rate": 4.620314165804964e-05, "loss": 0.0003341895353514701, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00033, "step": 300, "tokens/total": 9073936, "tokens/train_per_sec_per_gpu": 34.33, "tokens/trainable": 137411} +{"epoch": 1.17578125, "grad_norm": 0.02300048992037773, "learning_rate": 4.591863648168407e-05, "loss": 0.0001898434857139364, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00019, "step": 301, "tokens/total": 9104416, "tokens/train_per_sec_per_gpu": 35.79, "tokens/trainable": 137906} +{"epoch": 1.1796875, "grad_norm": 0.0014230191009119153, "learning_rate": 4.5634509217923135e-05, "loss": 2.900467188737821e-05, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00003, "step": 302, "tokens/total": 9134688, "tokens/train_per_sec_per_gpu": 34.33, "tokens/trainable": 138382} +{"epoch": 1.18359375, "grad_norm": 0.01370098628103733, "learning_rate": 4.535077169046201e-05, "loss": 0.0001739814761094749, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00017, "step": 303, "tokens/total": 9165088, "tokens/train_per_sec_per_gpu": 36.48, "tokens/trainable": 138841} +{"epoch": 1.1875, "grad_norm": 0.12571607530117035, "learning_rate": 4.506743570677743e-05, "loss": 0.00018397449457552284, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00018, "step": 304, "tokens/total": 9195584, "tokens/train_per_sec_per_gpu": 30.72, "tokens/trainable": 139268} +{"epoch": 1.19140625, "grad_norm": 0.2337377369403839, "learning_rate": 4.478451305763618e-05, "loss": 0.003163608256727457, "memory/device_reserved (GiB)": 35.67, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00317, "step": 305, "tokens/total": 9225760, "tokens/train_per_sec_per_gpu": 34.45, "tokens/trainable": 139735} +{"epoch": 1.1953125, "grad_norm": 0.029297636821866035, "learning_rate": 4.450201551660454e-05, "loss": 0.00035399867920204997, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00035, "step": 306, "tokens/total": 9256352, "tokens/train_per_sec_per_gpu": 37.06, "tokens/trainable": 140225} +{"epoch": 1.19921875, "grad_norm": 0.07336489856243134, "learning_rate": 4.4219954839558276e-05, "loss": 0.0006284696282818913, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.19, "memory/max_allocated (GiB)": 33.19, "ppl": 1.00063, "step": 307, "tokens/total": 9284256, "tokens/train_per_sec_per_gpu": 34.04, "tokens/trainable": 140663} +{"epoch": 1.203125, "grad_norm": 0.0036061827559024096, "learning_rate": 4.393834276419352e-05, "loss": 3.499729427858256e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00003, "step": 308, "tokens/total": 9314768, "tokens/train_per_sec_per_gpu": 29.94, "tokens/trainable": 141114} +{"epoch": 1.20703125, "grad_norm": 0.03760785236954689, "learning_rate": 4.36571910095382e-05, "loss": 0.0002682818449102342, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00027, "step": 309, "tokens/total": 9344976, "tokens/train_per_sec_per_gpu": 28.93, "tokens/trainable": 141529} +{"epoch": 1.2109375, "grad_norm": 0.005482436623424292, "learning_rate": 4.337651127546448e-05, "loss": 7.971160812303424e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00008, "step": 310, "tokens/total": 9375280, "tokens/train_per_sec_per_gpu": 34.27, "tokens/trainable": 141963} +{"epoch": 1.21484375, "grad_norm": 0.006238086149096489, "learning_rate": 4.3096315242201736e-05, "loss": 4.3831034417962655e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00004, "step": 311, "tokens/total": 9405536, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 142427} +{"epoch": 1.21875, "grad_norm": 0.00502822594717145, "learning_rate": 4.2816614569850635e-05, "loss": 5.458533996716142e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00005, "step": 312, "tokens/total": 9435696, "tokens/train_per_sec_per_gpu": 38.17, "tokens/trainable": 142893} +{"epoch": 1.22265625, "grad_norm": 0.17115379869937897, "learning_rate": 4.2537420897897864e-05, "loss": 0.0009321674006059766, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00093, "step": 313, "tokens/total": 9466080, "tokens/train_per_sec_per_gpu": 36.45, "tokens/trainable": 143367} +{"epoch": 1.2265625, "grad_norm": 0.009989157319068909, "learning_rate": 4.225874584473174e-05, "loss": 0.00014035131607670337, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00014, "step": 314, "tokens/total": 9496400, "tokens/train_per_sec_per_gpu": 32.97, "tokens/trainable": 143823} +{"epoch": 1.23046875, "grad_norm": 0.006546014454215765, "learning_rate": 4.19806010071587e-05, "loss": 7.929251296445727e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00008, "step": 315, "tokens/total": 9526880, "tokens/train_per_sec_per_gpu": 34.63, "tokens/trainable": 144305} +{"epoch": 1.234375, "grad_norm": 0.017240718007087708, "learning_rate": 4.170299795992081e-05, "loss": 0.0001234864175785333, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00012, "step": 316, "tokens/total": 9557280, "tokens/train_per_sec_per_gpu": 35.53, "tokens/trainable": 144762} +{"epoch": 1.23828125, "grad_norm": 0.011387556791305542, "learning_rate": 4.142594825521398e-05, "loss": 8.356192847713828e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00008, "step": 317, "tokens/total": 9587712, "tokens/train_per_sec_per_gpu": 37.5, "tokens/trainable": 145254} +{"epoch": 1.2421875, "grad_norm": 0.007485110778361559, "learning_rate": 4.114946342220728e-05, "loss": 8.404521213378757e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00008, "step": 318, "tokens/total": 9618064, "tokens/train_per_sec_per_gpu": 36.8, "tokens/trainable": 145713} +{"epoch": 1.24609375, "grad_norm": 0.24612121284008026, "learning_rate": 4.087355496656321e-05, "loss": 0.0037109816912561655, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00372, "step": 319, "tokens/total": 9648336, "tokens/train_per_sec_per_gpu": 36.28, "tokens/trainable": 146200} +{"epoch": 1.25, "grad_norm": 0.00244643772020936, "learning_rate": 4.05982343699588e-05, "loss": 4.710642315330915e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00005, "step": 320, "tokens/total": 9678688, "tokens/train_per_sec_per_gpu": 34.39, "tokens/trainable": 146668} +{"epoch": 1.25390625, "grad_norm": 0.018739959225058556, "learning_rate": 4.0323513089607876e-05, "loss": 0.00018197213648818433, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00018, "step": 321, "tokens/total": 9708848, "tokens/train_per_sec_per_gpu": 35.31, "tokens/trainable": 147114} +{"epoch": 1.2578125, "grad_norm": 0.22725771367549896, "learning_rate": 4.004940255778431e-05, "loss": 0.002088680863380432, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00209, "step": 322, "tokens/total": 9739360, "tokens/train_per_sec_per_gpu": 35.3, "tokens/trainable": 147564} +{"epoch": 1.26171875, "grad_norm": 0.000548983458429575, "learning_rate": 3.977591418134619e-05, "loss": 1.6327385310432874e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00002, "step": 323, "tokens/total": 9769776, "tokens/train_per_sec_per_gpu": 34.31, "tokens/trainable": 148013} +{"epoch": 1.265625, "grad_norm": 0.00761087192222476, "learning_rate": 3.95030593412612e-05, "loss": 5.408083234215155e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.01, "memory/max_allocated (GiB)": 34.01, "ppl": 1.00005, "step": 324, "tokens/total": 9800096, "tokens/train_per_sec_per_gpu": 38.23, "tokens/trainable": 148470} +{"epoch": 1.26953125, "grad_norm": 0.008723607286810875, "learning_rate": 3.923084939213296e-05, "loss": 8.67105700308457e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00009, "step": 325, "tokens/total": 9830304, "tokens/train_per_sec_per_gpu": 36.62, "tokens/trainable": 148926} +{"epoch": 1.2734375, "grad_norm": 0.006354155018925667, "learning_rate": 3.895929566172861e-05, "loss": 5.4418724175775424e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00005, "step": 326, "tokens/total": 9860608, "tokens/train_per_sec_per_gpu": 37.9, "tokens/trainable": 149420} +{"epoch": 1.27734375, "grad_norm": 0.7132415771484375, "learning_rate": 3.868840945050728e-05, "loss": 0.008354030549526215, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00839, "step": 327, "tokens/total": 9890560, "tokens/train_per_sec_per_gpu": 39.57, "tokens/trainable": 149873} +{"epoch": 1.28125, "grad_norm": 0.0944104865193367, "learning_rate": 3.841820203114995e-05, "loss": 0.0006982197519391775, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0007, "step": 328, "tokens/total": 9920880, "tokens/train_per_sec_per_gpu": 33.96, "tokens/trainable": 150347} +{"epoch": 1.28515625, "grad_norm": 0.0028970104176551104, "learning_rate": 3.814868464809027e-05, "loss": 3.3767173590604216e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00003, "step": 329, "tokens/total": 9951280, "tokens/train_per_sec_per_gpu": 31.93, "tokens/trainable": 150822} +{"epoch": 1.2890625, "grad_norm": 0.009064619429409504, "learning_rate": 3.787986851704667e-05, "loss": 7.040683703962713e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00007, "step": 330, "tokens/total": 9981600, "tokens/train_per_sec_per_gpu": 35.26, "tokens/trainable": 151266} +{"epoch": 1.29296875, "grad_norm": 0.05380578711628914, "learning_rate": 3.7611764824555654e-05, "loss": 0.00042552349623292685, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00043, "step": 331, "tokens/total": 10012016, "tokens/train_per_sec_per_gpu": 37.23, "tokens/trainable": 151734} +{"epoch": 1.296875, "grad_norm": 0.00875640194863081, "learning_rate": 3.734438472750619e-05, "loss": 6.885632319608703e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00007, "step": 332, "tokens/total": 10042448, "tokens/train_per_sec_per_gpu": 34.15, "tokens/trainable": 152184} +{"epoch": 1.30078125, "grad_norm": 0.0009642363293096423, "learning_rate": 3.707773935267552e-05, "loss": 2.1829437173437327e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00002, "step": 333, "tokens/total": 10073024, "tokens/train_per_sec_per_gpu": 32.0, "tokens/trainable": 152623} +{"epoch": 1.3046875, "grad_norm": 0.0004874366568401456, "learning_rate": 3.68118397962661e-05, "loss": 1.1361282304278575e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00001, "step": 334, "tokens/total": 10103504, "tokens/train_per_sec_per_gpu": 33.54, "tokens/trainable": 153040} +{"epoch": 1.30859375, "grad_norm": 0.31265145540237427, "learning_rate": 3.654669712344384e-05, "loss": 0.009118539281189442, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.31, "memory/max_allocated (GiB)": 33.31, "ppl": 1.00916, "step": 335, "tokens/total": 10131632, "tokens/train_per_sec_per_gpu": 31.81, "tokens/trainable": 153463} +{"epoch": 1.3125, "grad_norm": 0.0011828432325273752, "learning_rate": 3.628232236787763e-05, "loss": 2.4649223632877693e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00002, "step": 336, "tokens/total": 10161824, "tokens/train_per_sec_per_gpu": 33.41, "tokens/trainable": 153908} +{"epoch": 1.31640625, "grad_norm": 0.008298320695757866, "learning_rate": 3.6018726531280144e-05, "loss": 0.00010702587314881384, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00011, "step": 337, "tokens/total": 10192448, "tokens/train_per_sec_per_gpu": 33.84, "tokens/trainable": 154387} +{"epoch": 1.3203125, "grad_norm": 0.0037000810261815786, "learning_rate": 3.575592058295017e-05, "loss": 3.555983494152315e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00004, "step": 338, "tokens/total": 10222704, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 154836} +{"epoch": 1.32421875, "grad_norm": 0.30193620920181274, "learning_rate": 3.549391545931585e-05, "loss": 0.0002394289622316137, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00024, "step": 339, "tokens/total": 10253168, "tokens/train_per_sec_per_gpu": 29.94, "tokens/trainable": 155261} +{"epoch": 1.328125, "grad_norm": 0.013805638998746872, "learning_rate": 3.5232722063479914e-05, "loss": 9.105106437345967e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00009, "step": 340, "tokens/total": 10283552, "tokens/train_per_sec_per_gpu": 31.24, "tokens/trainable": 155674} +{"epoch": 1.33203125, "grad_norm": 0.26545384526252747, "learning_rate": 3.49723512647657e-05, "loss": 0.011088686995208263, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01115, "step": 341, "tokens/total": 10313872, "tokens/train_per_sec_per_gpu": 35.96, "tokens/trainable": 156164} +{"epoch": 1.3359375, "grad_norm": 0.00922760646790266, "learning_rate": 3.471281389826491e-05, "loss": 9.105133358389139e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00009, "step": 342, "tokens/total": 10344256, "tokens/train_per_sec_per_gpu": 33.75, "tokens/trainable": 156611} +{"epoch": 1.33984375, "grad_norm": 0.14466270804405212, "learning_rate": 3.4454120764386764e-05, "loss": 0.0014950309414416552, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.0015, "step": 343, "tokens/total": 10374544, "tokens/train_per_sec_per_gpu": 33.8, "tokens/trainable": 157073} +{"epoch": 1.34375, "grad_norm": 0.10482759773731232, "learning_rate": 3.4196282628408526e-05, "loss": 0.0009619007469154894, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00096, "step": 344, "tokens/total": 10405040, "tokens/train_per_sec_per_gpu": 33.1, "tokens/trainable": 157523} +{"epoch": 1.34765625, "grad_norm": 0.20004107058048248, "learning_rate": 3.3939310220027456e-05, "loss": 0.002495028544217348, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.0025, "step": 345, "tokens/total": 10435424, "tokens/train_per_sec_per_gpu": 35.16, "tokens/trainable": 157964} +{"epoch": 1.3515625, "grad_norm": 8.192997932434082, "learning_rate": 3.3683214232914404e-05, "loss": 0.006594266276806593, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00662, "step": 346, "tokens/total": 10465760, "tokens/train_per_sec_per_gpu": 30.54, "tokens/trainable": 158390} +{"epoch": 1.35546875, "grad_norm": 0.005605250597000122, "learning_rate": 3.342800532426873e-05, "loss": 6.323108391370624e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00006, "step": 347, "tokens/total": 10495904, "tokens/train_per_sec_per_gpu": 35.76, "tokens/trainable": 158848} +{"epoch": 1.359375, "grad_norm": 0.003419809276238084, "learning_rate": 3.317369411437484e-05, "loss": 5.6915632740128785e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.0, "memory/max_allocated (GiB)": 34.0, "ppl": 1.00006, "step": 348, "tokens/total": 10526624, "tokens/train_per_sec_per_gpu": 28.85, "tokens/trainable": 159270} +{"epoch": 1.36328125, "grad_norm": 0.18140868842601776, "learning_rate": 3.292029118616024e-05, "loss": 0.0037076647859066725, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00371, "step": 349, "tokens/total": 10556752, "tokens/train_per_sec_per_gpu": 32.67, "tokens/trainable": 159720} +{"epoch": 1.3671875, "grad_norm": 0.002614055061712861, "learning_rate": 3.266780708475511e-05, "loss": 3.247601125622168e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00003, "step": 350, "tokens/total": 10587264, "tokens/train_per_sec_per_gpu": 35.39, "tokens/trainable": 160187} +{"epoch": 1.37109375, "grad_norm": 0.002343076979741454, "learning_rate": 3.241625231705354e-05, "loss": 4.7991154133342206e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00005, "step": 351, "tokens/total": 10617504, "tokens/train_per_sec_per_gpu": 32.81, "tokens/trainable": 160645} +{"epoch": 1.375, "grad_norm": 0.03407059609889984, "learning_rate": 3.216563735127618e-05, "loss": 0.0003904080658685416, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00039, "step": 352, "tokens/total": 10647760, "tokens/train_per_sec_per_gpu": 34.71, "tokens/trainable": 161102} +{"epoch": 1.37890625, "grad_norm": 0.012719057500362396, "learning_rate": 3.191597261653475e-05, "loss": 0.00012954325939062983, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00013, "step": 353, "tokens/total": 10678080, "tokens/train_per_sec_per_gpu": 29.06, "tokens/trainable": 161555} +{"epoch": 1.3828125, "grad_norm": 0.005359884351491928, "learning_rate": 3.166726850239794e-05, "loss": 8.441988029517233e-05, "memory/device_reserved (GiB)": 35.41, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00008, "step": 354, "tokens/total": 10708544, "tokens/train_per_sec_per_gpu": 36.95, "tokens/trainable": 162020} +{"epoch": 1.38671875, "grad_norm": 0.056369855999946594, "learning_rate": 3.141953535845912e-05, "loss": 0.00027662873617373407, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00028, "step": 355, "tokens/total": 10739024, "tokens/train_per_sec_per_gpu": 32.31, "tokens/trainable": 162448} +{"epoch": 1.390625, "grad_norm": 0.0013983466196805239, "learning_rate": 3.11727834939056e-05, "loss": 3.224632018827833e-05, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.61, "memory/max_allocated (GiB)": 33.61, "ppl": 1.00003, "step": 356, "tokens/total": 10769024, "tokens/train_per_sec_per_gpu": 28.48, "tokens/trainable": 162863} +{"epoch": 1.39453125, "grad_norm": 0.06720387935638428, "learning_rate": 3.092702317708967e-05, "loss": 0.0002829490404110402, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00028, "step": 357, "tokens/total": 10799328, "tokens/train_per_sec_per_gpu": 35.16, "tokens/trainable": 163318} +{"epoch": 1.3984375, "grad_norm": 0.01048748753964901, "learning_rate": 3.0682264635101276e-05, "loss": 0.00019201112445443869, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00019, "step": 358, "tokens/total": 10829488, "tokens/train_per_sec_per_gpu": 33.28, "tokens/trainable": 163767} +{"epoch": 1.40234375, "grad_norm": 0.11898642778396606, "learning_rate": 3.0438518053342407e-05, "loss": 0.0006578543689101934, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00066, "step": 359, "tokens/total": 10859936, "tokens/train_per_sec_per_gpu": 32.13, "tokens/trainable": 164208} +{"epoch": 1.40625, "grad_norm": 0.09823726862668991, "learning_rate": 3.0195793575103266e-05, "loss": 0.001951234764419496, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00195, "step": 360, "tokens/total": 10890400, "tokens/train_per_sec_per_gpu": 34.39, "tokens/trainable": 164655} +{"epoch": 1.41015625, "grad_norm": 0.37214013934135437, "learning_rate": 2.9954101301140146e-05, "loss": 0.003241142490878701, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00325, "step": 361, "tokens/total": 10920624, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 165100} +{"epoch": 1.4140625, "grad_norm": 0.033950626850128174, "learning_rate": 2.9713451289255123e-05, "loss": 0.00024406128795817494, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00024, "step": 362, "tokens/total": 10951136, "tokens/train_per_sec_per_gpu": 35.19, "tokens/trainable": 165587} +{"epoch": 1.41796875, "grad_norm": 0.11754101514816284, "learning_rate": 2.9473853553877484e-05, "loss": 0.0018770417664200068, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00188, "step": 363, "tokens/total": 10981344, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 166034} +{"epoch": 1.421875, "grad_norm": 0.012853973545134068, "learning_rate": 2.9235318065647e-05, "loss": 0.00021306828421074897, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00021, "step": 364, "tokens/total": 11011552, "tokens/train_per_sec_per_gpu": 34.3, "tokens/trainable": 166485} +{"epoch": 1.42578125, "grad_norm": 0.09332777559757233, "learning_rate": 2.8997854750998964e-05, "loss": 0.00026301087928004563, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00026, "step": 365, "tokens/total": 11041872, "tokens/train_per_sec_per_gpu": 36.56, "tokens/trainable": 166973} +{"epoch": 1.4296875, "grad_norm": 0.009343368001282215, "learning_rate": 2.8761473491751258e-05, "loss": 0.0001053380110533908, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00011, "step": 366, "tokens/total": 11072192, "tokens/train_per_sec_per_gpu": 29.32, "tokens/trainable": 167414} +{"epoch": 1.43359375, "grad_norm": 0.035727277398109436, "learning_rate": 2.8526184124692883e-05, "loss": 0.0005394808831624687, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00054, "step": 367, "tokens/total": 11102736, "tokens/train_per_sec_per_gpu": 30.4, "tokens/trainable": 167851} +{"epoch": 1.4375, "grad_norm": 0.08836446702480316, "learning_rate": 2.829199644117484e-05, "loss": 0.000713829998858273, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00071, "step": 368, "tokens/total": 11133056, "tokens/train_per_sec_per_gpu": 35.79, "tokens/trainable": 168294} +{"epoch": 1.44140625, "grad_norm": 0.006047180388122797, "learning_rate": 2.8058920186702553e-05, "loss": 8.545963646611199e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00009, "step": 369, "tokens/total": 11163568, "tokens/train_per_sec_per_gpu": 32.4, "tokens/trainable": 168769} +{"epoch": 1.4453125, "grad_norm": 0.21181856095790863, "learning_rate": 2.782696506053033e-05, "loss": 0.0027023768052458763, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00271, "step": 370, "tokens/total": 11193936, "tokens/train_per_sec_per_gpu": 37.04, "tokens/trainable": 169270} +{"epoch": 1.44921875, "grad_norm": 0.001994711346924305, "learning_rate": 2.7596140715257824e-05, "loss": 3.8951005990384147e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00004, "step": 371, "tokens/total": 11224480, "tokens/train_per_sec_per_gpu": 35.27, "tokens/trainable": 169711} +{"epoch": 1.453125, "grad_norm": 0.0069135697558522224, "learning_rate": 2.7366456756428184e-05, "loss": 0.00011239978630328551, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00011, "step": 372, "tokens/total": 11254912, "tokens/train_per_sec_per_gpu": 33.12, "tokens/trainable": 170185} +{"epoch": 1.45703125, "grad_norm": 0.3907967507839203, "learning_rate": 2.7137922742128486e-05, "loss": 0.0018560648895800114, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00186, "step": 373, "tokens/total": 11285104, "tokens/train_per_sec_per_gpu": 33.35, "tokens/trainable": 170650} +{"epoch": 1.4609375, "grad_norm": 0.0009331480250693858, "learning_rate": 2.691054818259188e-05, "loss": 2.2479640392703004e-05, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00002, "step": 374, "tokens/total": 11315600, "tokens/train_per_sec_per_gpu": 32.22, "tokens/trainable": 171126} +{"epoch": 1.46484375, "grad_norm": 0.14613144099712372, "learning_rate": 2.6684342539801933e-05, "loss": 0.0031683319248259068, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00317, "step": 375, "tokens/total": 11345776, "tokens/train_per_sec_per_gpu": 35.85, "tokens/trainable": 171600} +{"epoch": 1.46875, "grad_norm": 0.00874971691519022, "learning_rate": 2.645931522709877e-05, "loss": 7.25445497664623e-05, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00007, "step": 376, "tokens/total": 11376208, "tokens/train_per_sec_per_gpu": 28.3, "tokens/trainable": 172017} +{"epoch": 1.47265625, "grad_norm": 0.02506718970835209, "learning_rate": 2.6235475608787365e-05, "loss": 0.00010988111171172932, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00011, "step": 377, "tokens/total": 11406512, "tokens/train_per_sec_per_gpu": 36.17, "tokens/trainable": 172493} +{"epoch": 1.4765625, "grad_norm": 0.09279019385576248, "learning_rate": 2.6012832999747916e-05, "loss": 0.0005218038568273187, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00052, "step": 378, "tokens/total": 11436544, "tokens/train_per_sec_per_gpu": 34.55, "tokens/trainable": 172951} +{"epoch": 1.48046875, "grad_norm": 0.002128554042428732, "learning_rate": 2.579139666504821e-05, "loss": 3.309818930574693e-05, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00003, "step": 379, "tokens/total": 11467008, "tokens/train_per_sec_per_gpu": 35.94, "tokens/trainable": 173435} +{"epoch": 1.484375, "grad_norm": 0.08014458417892456, "learning_rate": 2.557117581955798e-05, "loss": 0.00045113274245522916, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00045, "step": 380, "tokens/total": 11497184, "tokens/train_per_sec_per_gpu": 30.65, "tokens/trainable": 173900} +{"epoch": 1.48828125, "grad_norm": 0.009996578097343445, "learning_rate": 2.5352179627565532e-05, "loss": 0.00013630215835291892, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00014, "step": 381, "tokens/total": 11527600, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 174371} +{"epoch": 1.4921875, "grad_norm": 0.0028510303236544132, "learning_rate": 2.5134417202396277e-05, "loss": 2.7343612600816414e-05, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00003, "step": 382, "tokens/total": 11557808, "tokens/train_per_sec_per_gpu": 32.95, "tokens/trainable": 174826} +{"epoch": 1.49609375, "grad_norm": 0.0006682629464194179, "learning_rate": 2.491789760603361e-05, "loss": 9.333229172625579e-06, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00001, "step": 383, "tokens/total": 11588352, "tokens/train_per_sec_per_gpu": 33.1, "tokens/trainable": 175276} +{"epoch": 1.5, "grad_norm": 0.0072454060427844524, "learning_rate": 2.4702629848741764e-05, "loss": 8.043196430662647e-05, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00008, "step": 384, "tokens/total": 11618640, "tokens/train_per_sec_per_gpu": 33.88, "tokens/trainable": 175720} +{"epoch": 1.50390625, "grad_norm": 0.011281410232186317, "learning_rate": 2.4488622888690785e-05, "loss": 8.32078221719712e-05, "memory/device_reserved (GiB)": 35.89, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00008, "step": 385, "tokens/total": 11649072, "tokens/train_per_sec_per_gpu": 35.4, "tokens/trainable": 176203} +{"epoch": 1.5078125, "grad_norm": 0.1868448704481125, "learning_rate": 2.427588563158384e-05, "loss": 0.0016654160572215915, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00167, "step": 386, "tokens/total": 11679552, "tokens/train_per_sec_per_gpu": 34.5, "tokens/trainable": 176651} +{"epoch": 1.51171875, "grad_norm": 0.0011122338473796844, "learning_rate": 2.406442693028651e-05, "loss": 2.4154323909897357e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00002, "step": 387, "tokens/total": 11709760, "tokens/train_per_sec_per_gpu": 32.89, "tokens/trainable": 177115} +{"epoch": 1.515625, "grad_norm": 0.0008903060806915164, "learning_rate": 2.3854255584458547e-05, "loss": 2.0764218788826838e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00002, "step": 388, "tokens/total": 11740288, "tokens/train_per_sec_per_gpu": 36.23, "tokens/trainable": 177601} +{"epoch": 1.51953125, "grad_norm": 0.004497055895626545, "learning_rate": 2.3645380340187508e-05, "loss": 3.050938539672643e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00003, "step": 389, "tokens/total": 11770592, "tokens/train_per_sec_per_gpu": 33.15, "tokens/trainable": 178050} +{"epoch": 1.5234375, "grad_norm": 0.004634706303477287, "learning_rate": 2.3437809889624914e-05, "loss": 4.90621714561712e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00005, "step": 390, "tokens/total": 11800752, "tokens/train_per_sec_per_gpu": 36.59, "tokens/trainable": 178531} +{"epoch": 1.52734375, "grad_norm": 0.00251275347545743, "learning_rate": 2.3231552870624487e-05, "loss": 4.33354580309242e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00004, "step": 391, "tokens/total": 11831360, "tokens/train_per_sec_per_gpu": 34.13, "tokens/trainable": 179000} +{"epoch": 1.53125, "grad_norm": 0.004285548347979784, "learning_rate": 2.3026617866382657e-05, "loss": 5.757250255555846e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00006, "step": 392, "tokens/total": 11861552, "tokens/train_per_sec_per_gpu": 33.21, "tokens/trainable": 179457} +{"epoch": 1.53515625, "grad_norm": 0.08757233619689941, "learning_rate": 2.2823013405081507e-05, "loss": 2.351473449380137e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00002, "step": 393, "tokens/total": 11891904, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 179890} +{"epoch": 1.5390625, "grad_norm": 0.00247983168810606, "learning_rate": 2.2620747959533722e-05, "loss": 4.2569590732455254e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00004, "step": 394, "tokens/total": 11922208, "tokens/train_per_sec_per_gpu": 33.99, "tokens/trainable": 180341} +{"epoch": 1.54296875, "grad_norm": 0.0006376878009177744, "learning_rate": 2.2419829946830123e-05, "loss": 1.580168100190349e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00002, "step": 395, "tokens/total": 11952672, "tokens/train_per_sec_per_gpu": 33.07, "tokens/trainable": 180762} +{"epoch": 1.546875, "grad_norm": 0.010959293693304062, "learning_rate": 2.2220267727989325e-05, "loss": 9.101699106395245e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00009, "step": 396, "tokens/total": 11983088, "tokens/train_per_sec_per_gpu": 35.3, "tokens/trainable": 181226} +{"epoch": 1.55078125, "grad_norm": 0.0010561353992670774, "learning_rate": 2.202206960760984e-05, "loss": 1.6030653569032438e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00002, "step": 397, "tokens/total": 12013488, "tokens/train_per_sec_per_gpu": 33.31, "tokens/trainable": 181714} +{"epoch": 1.5546875, "grad_norm": 0.025496836751699448, "learning_rate": 2.182524383352446e-05, "loss": 0.00019739707931876183, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0002, "step": 398, "tokens/total": 12043968, "tokens/train_per_sec_per_gpu": 35.57, "tokens/trainable": 182142} +{"epoch": 1.55859375, "grad_norm": 0.00516202999278903, "learning_rate": 2.1629798596457056e-05, "loss": 5.949653859715909e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00006, "step": 399, "tokens/total": 12074160, "tokens/train_per_sec_per_gpu": 34.96, "tokens/trainable": 182604} +{"epoch": 1.5625, "grad_norm": 0.0005809378926642239, "learning_rate": 2.1435742029681725e-05, "loss": 1.5215588973660488e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.67, "memory/max_allocated (GiB)": 33.67, "ppl": 1.00002, "step": 400, "tokens/total": 12104352, "tokens/train_per_sec_per_gpu": 25.36, "tokens/trainable": 183011} +{"epoch": 1.56640625, "grad_norm": 0.0006219783099368215, "learning_rate": 2.124308220868431e-05, "loss": 1.3315910109668039e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00001, "step": 401, "tokens/total": 12134240, "tokens/train_per_sec_per_gpu": 31.9, "tokens/trainable": 183459} +{"epoch": 1.5703125, "grad_norm": 0.0050822533667087555, "learning_rate": 2.105182715082638e-05, "loss": 2.660456084413454e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00003, "step": 402, "tokens/total": 12164480, "tokens/train_per_sec_per_gpu": 38.81, "tokens/trainable": 183959} +{"epoch": 1.57421875, "grad_norm": 0.004483669530600309, "learning_rate": 2.0861984815011552e-05, "loss": 4.43336321040988e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00004, "step": 403, "tokens/total": 12194640, "tokens/train_per_sec_per_gpu": 32.65, "tokens/trainable": 184431} +{"epoch": 1.578125, "grad_norm": 0.002029991941526532, "learning_rate": 2.0673563101354323e-05, "loss": 2.8533231670735404e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00003, "step": 404, "tokens/total": 12224960, "tokens/train_per_sec_per_gpu": 31.51, "tokens/trainable": 184867} +{"epoch": 1.58203125, "grad_norm": 0.0010850686812773347, "learning_rate": 2.0486569850851317e-05, "loss": 1.5834324585739523e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00002, "step": 405, "tokens/total": 12255008, "tokens/train_per_sec_per_gpu": 29.46, "tokens/trainable": 185300} +{"epoch": 1.5859375, "grad_norm": 0.22020931541919708, "learning_rate": 2.0301012845054956e-05, "loss": 0.00125799176748842, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00126, "step": 406, "tokens/total": 12285248, "tokens/train_per_sec_per_gpu": 31.57, "tokens/trainable": 185754} +{"epoch": 1.58984375, "grad_norm": 0.003032378386706114, "learning_rate": 2.011689980574966e-05, "loss": 1.8148441085941158e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00002, "step": 407, "tokens/total": 12315552, "tokens/train_per_sec_per_gpu": 36.9, "tokens/trainable": 186213} +{"epoch": 1.59375, "grad_norm": 0.01703350804746151, "learning_rate": 1.993423839463052e-05, "loss": 2.1378953533712775e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00002, "step": 408, "tokens/total": 12345904, "tokens/train_per_sec_per_gpu": 32.92, "tokens/trainable": 186659} +{"epoch": 1.59765625, "grad_norm": 0.0006738354568369687, "learning_rate": 1.975303621298445e-05, "loss": 1.730798976495862e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00002, "step": 409, "tokens/total": 12376336, "tokens/train_per_sec_per_gpu": 33.12, "tokens/trainable": 187083} +{"epoch": 1.6015625, "grad_norm": 0.0016640514368191361, "learning_rate": 1.957330080137385e-05, "loss": 1.8207503671874292e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00002, "step": 410, "tokens/total": 12406880, "tokens/train_per_sec_per_gpu": 35.0, "tokens/trainable": 187530} +{"epoch": 1.60546875, "grad_norm": 0.05258834734559059, "learning_rate": 1.9395039639322864e-05, "loss": 0.00014502773410640657, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00015, "step": 411, "tokens/total": 12437088, "tokens/train_per_sec_per_gpu": 34.38, "tokens/trainable": 187972} +{"epoch": 1.609375, "grad_norm": 0.0004455884627532214, "learning_rate": 1.9218260145006073e-05, "loss": 1.1770534911192954e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00001, "step": 412, "tokens/total": 12465440, "tokens/train_per_sec_per_gpu": 40.06, "tokens/trainable": 188417} +{"epoch": 1.61328125, "grad_norm": 0.13378198444843292, "learning_rate": 1.904296967493982e-05, "loss": 0.0007207246962934732, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00072, "step": 413, "tokens/total": 12495968, "tokens/train_per_sec_per_gpu": 31.66, "tokens/trainable": 188868} +{"epoch": 1.6171875, "grad_norm": 0.017833322286605835, "learning_rate": 1.8869175523676064e-05, "loss": 0.00016917653556447476, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00017, "step": 414, "tokens/total": 12526128, "tokens/train_per_sec_per_gpu": 37.1, "tokens/trainable": 189337} +{"epoch": 1.62109375, "grad_norm": 0.0006367648602463305, "learning_rate": 1.869688492349885e-05, "loss": 1.2300321031943895e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00001, "step": 415, "tokens/total": 12556256, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 189750} +{"epoch": 1.625, "grad_norm": 0.00034314877120777965, "learning_rate": 1.85261050441233e-05, "loss": 1.0386990652477834e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00001, "step": 416, "tokens/total": 12586736, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 190216} +{"epoch": 1.62890625, "grad_norm": 0.11333048343658447, "learning_rate": 1.8356842992397304e-05, "loss": 0.00011449151497799903, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00011, "step": 417, "tokens/total": 12617168, "tokens/train_per_sec_per_gpu": 35.18, "tokens/trainable": 190708} +{"epoch": 1.6328125, "grad_norm": 0.002743076765909791, "learning_rate": 1.8189105812005714e-05, "loss": 1.5843726941966452e-05, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00002, "step": 418, "tokens/total": 12647680, "tokens/train_per_sec_per_gpu": 30.93, "tokens/trainable": 191126} +{"epoch": 1.63671875, "grad_norm": 0.0011538882972672582, "learning_rate": 1.802290048317732e-05, "loss": 1.4711402400280349e-05, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00001, "step": 419, "tokens/total": 12677952, "tokens/train_per_sec_per_gpu": 30.43, "tokens/trainable": 191540} +{"epoch": 1.640625, "grad_norm": 0.004467409569770098, "learning_rate": 1.785823392239424e-05, "loss": 4.823958806809969e-05, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00005, "step": 420, "tokens/total": 12708416, "tokens/train_per_sec_per_gpu": 36.79, "tokens/trainable": 192014} +{"epoch": 1.64453125, "grad_norm": 0.0012163098435848951, "learning_rate": 1.7695112982104225e-05, "loss": 1.1310569789202418e-05, "memory/device_reserved (GiB)": 35.8, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00001, "step": 421, "tokens/total": 12738640, "tokens/train_per_sec_per_gpu": 31.42, "tokens/trainable": 192463} +{"epoch": 1.6484375, "grad_norm": 0.05586014315485954, "learning_rate": 1.7533544450435433e-05, "loss": 0.0002712146961130202, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00027, "step": 422, "tokens/total": 12769232, "tokens/train_per_sec_per_gpu": 32.82, "tokens/trainable": 192918} +{"epoch": 1.65234375, "grad_norm": 0.06670165061950684, "learning_rate": 1.7373535050913946e-05, "loss": 0.0004127066058572382, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00041, "step": 423, "tokens/total": 12799648, "tokens/train_per_sec_per_gpu": 37.51, "tokens/trainable": 193425} +{"epoch": 1.65625, "grad_norm": 0.019891072064638138, "learning_rate": 1.721509144218405e-05, "loss": 0.00023232161765918136, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00023, "step": 424, "tokens/total": 12829920, "tokens/train_per_sec_per_gpu": 35.36, "tokens/trainable": 193898} +{"epoch": 1.66015625, "grad_norm": 0.006559237837791443, "learning_rate": 1.705822021773101e-05, "loss": 6.114803545642644e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00006, "step": 425, "tokens/total": 12860112, "tokens/train_per_sec_per_gpu": 29.38, "tokens/trainable": 194315} +{"epoch": 1.6640625, "grad_norm": 0.005621429067105055, "learning_rate": 1.69029279056068e-05, "loss": 6.088989175623283e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00006, "step": 426, "tokens/total": 12890320, "tokens/train_per_sec_per_gpu": 35.06, "tokens/trainable": 194796} +{"epoch": 1.66796875, "grad_norm": 0.0004855124861933291, "learning_rate": 1.6749220968158415e-05, "loss": 1.0911244316957891e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00001, "step": 427, "tokens/total": 12920656, "tokens/train_per_sec_per_gpu": 34.07, "tokens/trainable": 195262} +{"epoch": 1.671875, "grad_norm": 0.0006959444144740701, "learning_rate": 1.659710580175893e-05, "loss": 1.4198772987583652e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00001, "step": 428, "tokens/total": 12951040, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 195680} +{"epoch": 1.67578125, "grad_norm": 0.00048281854833476245, "learning_rate": 1.644658873654133e-05, "loss": 1.0396102879894897e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00001, "step": 429, "tokens/total": 12981232, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 196145} +{"epoch": 1.6796875, "grad_norm": 0.007472805213183165, "learning_rate": 1.629767603613508e-05, "loss": 5.188406430534087e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00005, "step": 430, "tokens/total": 13011616, "tokens/train_per_sec_per_gpu": 32.5, "tokens/trainable": 196626} +{"epoch": 1.68359375, "grad_norm": 0.008452537469565868, "learning_rate": 1.615037389740547e-05, "loss": 2.078613033518195e-05, "memory/device_reserved (GiB)": 36.35, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.00002, "step": 431, "tokens/total": 13042208, "tokens/train_per_sec_per_gpu": 31.22, "tokens/trainable": 197067} +{"epoch": 1.6875, "grad_norm": 0.0006947954534552991, "learning_rate": 1.600468845019576e-05, "loss": 1.0801179087138735e-05, "memory/device_reserved (GiB)": 36.35, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00001, "step": 432, "tokens/total": 13072800, "tokens/train_per_sec_per_gpu": 38.63, "tokens/trainable": 197565} +{"epoch": 1.69140625, "grad_norm": 0.0020175843965262175, "learning_rate": 1.5860625757072092e-05, "loss": 1.564873855386395e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00002, "step": 433, "tokens/total": 13103328, "tokens/train_per_sec_per_gpu": 33.04, "tokens/trainable": 198015} +{"epoch": 1.6953125, "grad_norm": 0.001340522081591189, "learning_rate": 1.571819181307116e-05, "loss": 1.7318390746368095e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00002, "step": 434, "tokens/total": 13133472, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 198442} +{"epoch": 1.69921875, "grad_norm": 0.08985895663499832, "learning_rate": 1.557739254545075e-05, "loss": 0.0004795463755726814, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00048, "step": 435, "tokens/total": 13164064, "tokens/train_per_sec_per_gpu": 34.45, "tokens/trainable": 198914} +{"epoch": 1.703125, "grad_norm": 0.0005816713673993945, "learning_rate": 1.543823381344311e-05, "loss": 1.203996089316206e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00001, "step": 436, "tokens/total": 13194464, "tokens/train_per_sec_per_gpu": 33.56, "tokens/trainable": 199355} +{"epoch": 1.70703125, "grad_norm": 0.2994021773338318, "learning_rate": 1.5300721408011114e-05, "loss": 0.0055721537210047245, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00559, "step": 437, "tokens/total": 13224992, "tokens/train_per_sec_per_gpu": 32.98, "tokens/trainable": 199830} +{"epoch": 1.7109375, "grad_norm": 0.0006708212895318866, "learning_rate": 1.5164861051607254e-05, "loss": 9.834290722210426e-06, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00001, "step": 438, "tokens/total": 13255296, "tokens/train_per_sec_per_gpu": 35.03, "tokens/trainable": 200338} +{"epoch": 1.71484375, "grad_norm": 0.004102411679923534, "learning_rate": 1.5030658397935521e-05, "loss": 2.766325997072272e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00003, "step": 439, "tokens/total": 13285344, "tokens/train_per_sec_per_gpu": 32.91, "tokens/trainable": 200785} +{"epoch": 1.71875, "grad_norm": 0.0009204484522342682, "learning_rate": 1.4898119031716104e-05, "loss": 1.4549179468303919e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00001, "step": 440, "tokens/total": 13315632, "tokens/train_per_sec_per_gpu": 36.38, "tokens/trainable": 201267} +{"epoch": 1.72265625, "grad_norm": 0.0014299805043265224, "learning_rate": 1.476724846845306e-05, "loss": 2.365603722864762e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00002, "step": 441, "tokens/total": 13346048, "tokens/train_per_sec_per_gpu": 34.22, "tokens/trainable": 201734} +{"epoch": 1.7265625, "grad_norm": 0.01532546803355217, "learning_rate": 1.463805215420471e-05, "loss": 9.99852636596188e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0001, "step": 442, "tokens/total": 13376304, "tokens/train_per_sec_per_gpu": 30.36, "tokens/trainable": 202174} +{"epoch": 1.73046875, "grad_norm": 0.18507054448127747, "learning_rate": 1.451053546535705e-05, "loss": 0.000987955485470593, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00099, "step": 443, "tokens/total": 13406720, "tokens/train_per_sec_per_gpu": 36.23, "tokens/trainable": 202612} +{"epoch": 1.734375, "grad_norm": 0.028212063014507294, "learning_rate": 1.438470370840001e-05, "loss": 0.00025927621754817665, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.64, "memory/max_allocated (GiB)": 33.64, "ppl": 1.00026, "step": 444, "tokens/total": 13436464, "tokens/train_per_sec_per_gpu": 30.96, "tokens/trainable": 203020} +{"epoch": 1.73828125, "grad_norm": 0.0004877297324128449, "learning_rate": 1.4260562119706606e-05, "loss": 1.2777243682648987e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00001, "step": 445, "tokens/total": 13466672, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 203479} +{"epoch": 1.7421875, "grad_norm": 0.02534927800297737, "learning_rate": 1.413811586531508e-05, "loss": 3.3006788726197556e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00003, "step": 446, "tokens/total": 13496736, "tokens/train_per_sec_per_gpu": 35.38, "tokens/trainable": 203946} +{"epoch": 1.74609375, "grad_norm": 0.0006437217816710472, "learning_rate": 1.4017370040713884e-05, "loss": 1.1569853086257353e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00001, "step": 447, "tokens/total": 13526928, "tokens/train_per_sec_per_gpu": 37.38, "tokens/trainable": 204382} +{"epoch": 1.75, "grad_norm": 0.0014389768475666642, "learning_rate": 1.3898329670629645e-05, "loss": 2.1063802705612034e-05, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00002, "step": 448, "tokens/total": 13557392, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 204830} +{"epoch": 1.75390625, "grad_norm": 0.016985442489385605, "learning_rate": 1.3780999708818058e-05, "loss": 0.00012392934877425432, "memory/device_reserved (GiB)": 36.8, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00012, "step": 449, "tokens/total": 13587776, "tokens/train_per_sec_per_gpu": 33.74, "tokens/trainable": 205300} +{"epoch": 1.7578125, "grad_norm": 0.008965734392404556, "learning_rate": 1.3665385037857758e-05, "loss": 6.039683285052888e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00006, "step": 450, "tokens/total": 13618112, "tokens/train_per_sec_per_gpu": 30.72, "tokens/trainable": 205726} +{"epoch": 1.76171875, "grad_norm": 0.5729678273200989, "learning_rate": 1.3551490468947126e-05, "loss": 0.00580303929746151, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00582, "step": 451, "tokens/total": 13648512, "tokens/train_per_sec_per_gpu": 31.74, "tokens/trainable": 206163} +{"epoch": 1.765625, "grad_norm": 0.0277901329100132, "learning_rate": 1.3439320741704075e-05, "loss": 9.134741412708536e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00009, "step": 452, "tokens/total": 13678704, "tokens/train_per_sec_per_gpu": 35.58, "tokens/trainable": 206619} +{"epoch": 1.76953125, "grad_norm": 0.3788454532623291, "learning_rate": 1.3328880523968808e-05, "loss": 0.002095964504405856, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.0021, "step": 453, "tokens/total": 13709232, "tokens/train_per_sec_per_gpu": 37.33, "tokens/trainable": 207101} +{"epoch": 1.7734375, "grad_norm": 0.00038695961120538414, "learning_rate": 1.3220174411609587e-05, "loss": 9.780245818546973e-06, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00001, "step": 454, "tokens/total": 13739600, "tokens/train_per_sec_per_gpu": 35.82, "tokens/trainable": 207546} +{"epoch": 1.77734375, "grad_norm": 0.01514297816902399, "learning_rate": 1.3113206928331471e-05, "loss": 7.800674939062446e-05, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00008, "step": 455, "tokens/total": 13769936, "tokens/train_per_sec_per_gpu": 34.44, "tokens/trainable": 207989} +{"epoch": 1.78125, "grad_norm": 0.0018013569060713053, "learning_rate": 1.300798252548806e-05, "loss": 2.1191644918872043e-05, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00002, "step": 456, "tokens/total": 13800240, "tokens/train_per_sec_per_gpu": 34.06, "tokens/trainable": 208402} +{"epoch": 1.78515625, "grad_norm": 0.0006610782584175467, "learning_rate": 1.2904505581896265e-05, "loss": 1.3429066711978521e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00001, "step": 457, "tokens/total": 13830512, "tokens/train_per_sec_per_gpu": 36.81, "tokens/trainable": 208904} +{"epoch": 1.7890625, "grad_norm": 0.0005749509437009692, "learning_rate": 1.2802780403654082e-05, "loss": 1.3295277312863618e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00001, "step": 458, "tokens/total": 13860832, "tokens/train_per_sec_per_gpu": 33.86, "tokens/trainable": 209382} +{"epoch": 1.79296875, "grad_norm": 0.00635139923542738, "learning_rate": 1.2702811223961408e-05, "loss": 5.09147321281489e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00005, "step": 459, "tokens/total": 13890992, "tokens/train_per_sec_per_gpu": 34.01, "tokens/trainable": 209833} +{"epoch": 1.796875, "grad_norm": 0.013369702734053135, "learning_rate": 1.2604602202943861e-05, "loss": 0.00011205296323169023, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00011, "step": 460, "tokens/total": 13921424, "tokens/train_per_sec_per_gpu": 32.79, "tokens/trainable": 210285} +{"epoch": 1.80078125, "grad_norm": 0.016762293875217438, "learning_rate": 1.2508157427479686e-05, "loss": 0.00019885516667272896, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.66, "memory/max_allocated (GiB)": 33.66, "ppl": 1.0002, "step": 461, "tokens/total": 13951504, "tokens/train_per_sec_per_gpu": 34.23, "tokens/trainable": 210770} +{"epoch": 1.8046875, "grad_norm": 0.006627500057220459, "learning_rate": 1.2413480911029655e-05, "loss": 2.5290853955084458e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.38, "memory/max_allocated (GiB)": 33.38, "ppl": 1.00003, "step": 462, "tokens/total": 13979728, "tokens/train_per_sec_per_gpu": 30.5, "tokens/trainable": 211194} +{"epoch": 1.80859375, "grad_norm": 0.003884925739839673, "learning_rate": 1.2320576593470082e-05, "loss": 2.766420948319137e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00003, "step": 463, "tokens/total": 14009936, "tokens/train_per_sec_per_gpu": 35.34, "tokens/trainable": 211656} +{"epoch": 1.8125, "grad_norm": 0.000377490563550964, "learning_rate": 1.2229448340928828e-05, "loss": 9.118146408582106e-06, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00001, "step": 464, "tokens/total": 14039872, "tokens/train_per_sec_per_gpu": 29.49, "tokens/trainable": 212086} +{"epoch": 1.81640625, "grad_norm": 0.01634475402534008, "learning_rate": 1.2140099945624458e-05, "loss": 7.731329969828948e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00008, "step": 465, "tokens/total": 14070096, "tokens/train_per_sec_per_gpu": 32.84, "tokens/trainable": 212571} +{"epoch": 1.8203125, "grad_norm": 0.5773619413375854, "learning_rate": 1.205253512570841e-05, "loss": 0.008904650807380676, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00894, "step": 466, "tokens/total": 14100192, "tokens/train_per_sec_per_gpu": 33.09, "tokens/trainable": 213010} +{"epoch": 1.82421875, "grad_norm": 0.005223860964179039, "learning_rate": 1.1966757525110255e-05, "loss": 3.063467738684267e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00003, "step": 467, "tokens/total": 14130448, "tokens/train_per_sec_per_gpu": 31.72, "tokens/trainable": 213449} +{"epoch": 1.828125, "grad_norm": 0.00045610740198753774, "learning_rate": 1.1882770713386095e-05, "loss": 1.0150557500310242e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00001, "step": 468, "tokens/total": 14160480, "tokens/train_per_sec_per_gpu": 35.84, "tokens/trainable": 213897} +{"epoch": 1.83203125, "grad_norm": 0.005935850087553263, "learning_rate": 1.180057818556998e-05, "loss": 4.396100121084601e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00004, "step": 469, "tokens/total": 14191152, "tokens/train_per_sec_per_gpu": 37.02, "tokens/trainable": 214365} +{"epoch": 1.8359375, "grad_norm": 0.0010427028173580766, "learning_rate": 1.1720183362028494e-05, "loss": 2.14513493119739e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00002, "step": 470, "tokens/total": 14219392, "tokens/train_per_sec_per_gpu": 35.86, "tokens/trainable": 214799} +{"epoch": 1.83984375, "grad_norm": 0.006076959893107414, "learning_rate": 1.1641589588318387e-05, "loss": 1.3421818948700093e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00001, "step": 471, "tokens/total": 14249920, "tokens/train_per_sec_per_gpu": 35.62, "tokens/trainable": 215291} +{"epoch": 1.84375, "grad_norm": 0.0019434966379776597, "learning_rate": 1.1564800135047418e-05, "loss": 3.379978079465218e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00003, "step": 472, "tokens/total": 14280048, "tokens/train_per_sec_per_gpu": 34.1, "tokens/trainable": 215740} +{"epoch": 1.84765625, "grad_norm": 0.01917141303420067, "learning_rate": 1.148981819773816e-05, "loss": 8.30673161544837e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00008, "step": 473, "tokens/total": 14310432, "tokens/train_per_sec_per_gpu": 34.88, "tokens/trainable": 216193} +{"epoch": 1.8515625, "grad_norm": 0.027403976768255234, "learning_rate": 1.1416646896695086e-05, "loss": 5.644366319756955e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00006, "step": 474, "tokens/total": 14340496, "tokens/train_per_sec_per_gpu": 33.46, "tokens/trainable": 216633} +{"epoch": 1.85546875, "grad_norm": 0.0030669430270791054, "learning_rate": 1.1345289276874717e-05, "loss": 3.4152639273088425e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00003, "step": 475, "tokens/total": 14370704, "tokens/train_per_sec_per_gpu": 30.62, "tokens/trainable": 217039} +{"epoch": 1.859375, "grad_norm": 0.0022093369625508785, "learning_rate": 1.1275748307758873e-05, "loss": 3.003427991643548e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00003, "step": 476, "tokens/total": 14401120, "tokens/train_per_sec_per_gpu": 37.85, "tokens/trainable": 217529} +{"epoch": 1.86328125, "grad_norm": 0.007151913829147816, "learning_rate": 1.1208026883231147e-05, "loss": 4.787252328242175e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00005, "step": 477, "tokens/total": 14431344, "tokens/train_per_sec_per_gpu": 34.7, "tokens/trainable": 217987} +{"epoch": 1.8671875, "grad_norm": 0.005195859353989363, "learning_rate": 1.1142127821456433e-05, "loss": 6.279069930315018e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00006, "step": 478, "tokens/total": 14461952, "tokens/train_per_sec_per_gpu": 32.73, "tokens/trainable": 218460} +{"epoch": 1.87109375, "grad_norm": 0.0024305693805217743, "learning_rate": 1.1078053864763674e-05, "loss": 4.7390385589096695e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00005, "step": 479, "tokens/total": 14491968, "tokens/train_per_sec_per_gpu": 32.87, "tokens/trainable": 218908} +{"epoch": 1.875, "grad_norm": 0.0007113535539247096, "learning_rate": 1.1015807679531756e-05, "loss": 1.3227703675511293e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00001, "step": 480, "tokens/total": 14522528, "tokens/train_per_sec_per_gpu": 34.24, "tokens/trainable": 219349} +{"epoch": 1.87890625, "grad_norm": 0.0017267197836190462, "learning_rate": 1.0955391856078528e-05, "loss": 1.1387310223653913e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00001, "step": 481, "tokens/total": 14552544, "tokens/train_per_sec_per_gpu": 29.38, "tokens/trainable": 219775} +{"epoch": 1.8828125, "grad_norm": 0.01939094066619873, "learning_rate": 1.0896808908553007e-05, "loss": 7.642045238753781e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00008, "step": 482, "tokens/total": 14583040, "tokens/train_per_sec_per_gpu": 35.27, "tokens/trainable": 220283} +{"epoch": 1.88671875, "grad_norm": 0.004050788469612598, "learning_rate": 1.0840061274830763e-05, "loss": 5.199073348194361e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00005, "step": 483, "tokens/total": 14613136, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 220713} +{"epoch": 1.890625, "grad_norm": 0.0024211062118411064, "learning_rate": 1.0785151316412473e-05, "loss": 2.763515840342734e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00003, "step": 484, "tokens/total": 14643312, "tokens/train_per_sec_per_gpu": 35.18, "tokens/trainable": 221155} +{"epoch": 1.89453125, "grad_norm": 0.015418405644595623, "learning_rate": 1.0732081318325639e-05, "loss": 7.891400309745222e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00008, "step": 485, "tokens/total": 14673472, "tokens/train_per_sec_per_gpu": 31.78, "tokens/trainable": 221604} +{"epoch": 1.8984375, "grad_norm": 0.06489475816488266, "learning_rate": 1.0680853489029501e-05, "loss": 0.00017086212756112218, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00017, "step": 486, "tokens/total": 14703984, "tokens/train_per_sec_per_gpu": 30.42, "tokens/trainable": 222040} +{"epoch": 1.90234375, "grad_norm": 0.000592821161262691, "learning_rate": 1.0631469960323152e-05, "loss": 1.0421663318993524e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.67, "memory/max_allocated (GiB)": 33.67, "ppl": 1.00001, "step": 487, "tokens/total": 14733968, "tokens/train_per_sec_per_gpu": 32.16, "tokens/trainable": 222472} +{"epoch": 1.90625, "grad_norm": 0.00045594078255817294, "learning_rate": 1.0583932787256783e-05, "loss": 1.1962510143348482e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00001, "step": 488, "tokens/total": 14764384, "tokens/train_per_sec_per_gpu": 36.41, "tokens/trainable": 222965} +{"epoch": 1.91015625, "grad_norm": 0.001553925801999867, "learning_rate": 1.0538243948046206e-05, "loss": 2.3432628950104117e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00002, "step": 489, "tokens/total": 14794656, "tokens/train_per_sec_per_gpu": 31.93, "tokens/trainable": 223418} +{"epoch": 1.9140625, "grad_norm": 0.0029810869600623846, "learning_rate": 1.0494405343990523e-05, "loss": 2.7821279218187556e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00003, "step": 490, "tokens/total": 14824768, "tokens/train_per_sec_per_gpu": 29.52, "tokens/trainable": 223862} +{"epoch": 1.91796875, "grad_norm": 0.00159536674618721, "learning_rate": 1.0452418799392985e-05, "loss": 2.1142092009540647e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00002, "step": 491, "tokens/total": 14855040, "tokens/train_per_sec_per_gpu": 32.93, "tokens/trainable": 224319} +{"epoch": 1.921875, "grad_norm": 0.003593488596379757, "learning_rate": 1.0412286061485102e-05, "loss": 3.937885048799217e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.01, "memory/max_allocated (GiB)": 34.01, "ppl": 1.00004, "step": 492, "tokens/total": 14885376, "tokens/train_per_sec_per_gpu": 36.5, "tokens/trainable": 224800} +{"epoch": 1.92578125, "grad_norm": 0.0018238810589537024, "learning_rate": 1.03740088003539e-05, "loss": 2.2674561478197575e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00002, "step": 493, "tokens/total": 14915744, "tokens/train_per_sec_per_gpu": 34.9, "tokens/trainable": 225282} +{"epoch": 1.9296875, "grad_norm": 0.0027770590968430042, "learning_rate": 1.0337588608872463e-05, "loss": 1.480587525293231e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00001, "step": 494, "tokens/total": 14946176, "tokens/train_per_sec_per_gpu": 32.91, "tokens/trainable": 225731} +{"epoch": 1.93359375, "grad_norm": 0.03237319365143776, "learning_rate": 1.0303027002633622e-05, "loss": 0.00021609703253488988, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00022, "step": 495, "tokens/total": 14976256, "tokens/train_per_sec_per_gpu": 32.69, "tokens/trainable": 226170} +{"epoch": 1.9375, "grad_norm": 0.00870265532284975, "learning_rate": 1.0270325419886884e-05, "loss": 5.133711965754628e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00005, "step": 496, "tokens/total": 15006528, "tokens/train_per_sec_per_gpu": 34.97, "tokens/trainable": 226634} +{"epoch": 1.94140625, "grad_norm": 0.0011709899408742785, "learning_rate": 1.0239485221478599e-05, "loss": 2.115583083650563e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00002, "step": 497, "tokens/total": 15037024, "tokens/train_per_sec_per_gpu": 33.07, "tokens/trainable": 227065} +{"epoch": 1.9453125, "grad_norm": 0.0010721588041633368, "learning_rate": 1.0210507690795292e-05, "loss": 1.7287293303525075e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00002, "step": 498, "tokens/total": 15067328, "tokens/train_per_sec_per_gpu": 37.81, "tokens/trainable": 227530} +{"epoch": 1.94921875, "grad_norm": 0.06478072702884674, "learning_rate": 1.0183394033710305e-05, "loss": 0.000421134231146425, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00042, "step": 499, "tokens/total": 15097632, "tokens/train_per_sec_per_gpu": 36.76, "tokens/trainable": 228005} +{"epoch": 1.953125, "grad_norm": 0.01197084691375494, "learning_rate": 1.0158145378533583e-05, "loss": 0.00013856604346074164, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00014, "step": 500, "tokens/total": 15127984, "tokens/train_per_sec_per_gpu": 31.03, "tokens/trainable": 228455} +{"epoch": 1.95703125, "grad_norm": 0.0047052945010364056, "learning_rate": 1.0134762775964726e-05, "loss": 3.4345306630712e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00003, "step": 501, "tokens/total": 15158336, "tokens/train_per_sec_per_gpu": 35.36, "tokens/trainable": 228926} +{"epoch": 1.9609375, "grad_norm": 0.01203920878469944, "learning_rate": 1.0113247199049278e-05, "loss": 5.282826168695465e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.03, "memory/max_allocated (GiB)": 34.03, "ppl": 1.00005, "step": 502, "tokens/total": 15189120, "tokens/train_per_sec_per_gpu": 32.25, "tokens/trainable": 229366} +{"epoch": 1.96484375, "grad_norm": 0.06692500412464142, "learning_rate": 1.0093599543138205e-05, "loss": 3.876092523569241e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00004, "step": 503, "tokens/total": 15219600, "tokens/train_per_sec_per_gpu": 32.93, "tokens/trainable": 229813} +{"epoch": 1.96875, "grad_norm": 0.05963515117764473, "learning_rate": 1.0075820625850675e-05, "loss": 0.0003227375273127109, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00032, "step": 504, "tokens/total": 15250016, "tokens/train_per_sec_per_gpu": 39.03, "tokens/trainable": 230317} +{"epoch": 1.97265625, "grad_norm": 0.007018797565251589, "learning_rate": 1.0059911187040013e-05, "loss": 5.134383536642417e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00005, "step": 505, "tokens/total": 15280576, "tokens/train_per_sec_per_gpu": 37.69, "tokens/trainable": 230805} +{"epoch": 1.9765625, "grad_norm": 0.006387659814208746, "learning_rate": 1.0045871888762893e-05, "loss": 4.712470399681479e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00005, "step": 506, "tokens/total": 15310800, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 231255} +{"epoch": 1.98046875, "grad_norm": 0.001986650750041008, "learning_rate": 1.003370331525184e-05, "loss": 2.7338617655914277e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00003, "step": 507, "tokens/total": 15341408, "tokens/train_per_sec_per_gpu": 36.48, "tokens/trainable": 231744} +{"epoch": 1.984375, "grad_norm": 0.0114446384832263, "learning_rate": 1.002340597289085e-05, "loss": 3.070381353609264e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.33, "memory/max_allocated (GiB)": 33.33, "ppl": 1.00003, "step": 508, "tokens/total": 15369648, "tokens/train_per_sec_per_gpu": 33.13, "tokens/trainable": 232170} +{"epoch": 1.98828125, "grad_norm": 0.0035379640758037567, "learning_rate": 1.0014980290194387e-05, "loss": 2.992351073771715e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.00003, "step": 509, "tokens/total": 15400480, "tokens/train_per_sec_per_gpu": 37.24, "tokens/trainable": 232678} +{"epoch": 1.9921875, "grad_norm": 0.13356627523899078, "learning_rate": 1.0008426617789489e-05, "loss": 0.0014438352081924677, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00144, "step": 510, "tokens/total": 15430848, "tokens/train_per_sec_per_gpu": 36.2, "tokens/trainable": 233167} +{"epoch": 1.99609375, "grad_norm": 0.0027949470095336437, "learning_rate": 1.0003745228401215e-05, "loss": 1.4765095329494216e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00001, "step": 511, "tokens/total": 15461216, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 233607} +{"epoch": 2.0, "grad_norm": 0.328495055437088, "learning_rate": 1.0000936316841296e-05, "loss": 0.0018887810874730349, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00189, "step": 512, "tokens/total": 15491744, "tokens/train_per_sec_per_gpu": 36.34, "tokens/trainable": 234076}