diff --git a/.gitattributes b/.gitattributes index 754d14b751359813c2570fdd43ca50c7f5cd6cda..e7e5cce5e7e26c28f4a6768a4151d6807ce06ec7 100644 --- a/.gitattributes +++ b/.gitattributes @@ -375,3 +375,20 @@ aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-512/tokeniz aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text aft_wave_v2/coin_real_4x__charter0p2/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer.json filter=lfs diff=lfs merge=lfs -text +aft_wave_v2/control_matched__charter2/training/checkpoints/tokenizer.json filter=lfs diff=lfs merge=lfs -text diff --git a/aft_wave_v2/control_matched__charter2/training/ARTIFACT_MANIFEST.local.json b/aft_wave_v2/control_matched__charter2/training/ARTIFACT_MANIFEST.local.json new file mode 100644 index 0000000000000000000000000000000000000000..5e877d0d733ebf6b3cf49958098a66063cd9a4eb --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/ARTIFACT_MANIFEST.local.json @@ -0,0 +1,867 @@ +{ + "repo": "arcadia-impact/scimt-dispatch-models", + "remote_prefix": "aft_wave_v2/control_matched__charter2/training", + "local_folder": "/workspace/wave/training", + "files": { + "TRAINED.json": { + "size": 974, + "sha256": "7a0f38d533bed04f257b6f2e311d037c6773944ddacad833c2df15735e93a69a" + }, + "axolotl.yaml": { + "size": 1210, + "sha256": "85e1ac7519356eb24741e70e76c15262c684b41306bccddca8d2bf7f8925e98b" + }, + "checkpoint.json": { + "size": 2234, + "sha256": "a194a3f99ae4570f38fa6781fc811575a122a934fbf2d2af71b1cee9f889bd7e" + }, + "checkpoints/README.md": { + "size": 2934, + "sha256": "ba0ccabd45078aa61d52a6c0ba4381c7a96d7197c492a9eb00693a0898980cc6" + }, + "checkpoints/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/adapter_model.safetensors": { + "size": 547777976, + "sha256": "6f7da5abb2cf5cc22da25374898331b609fa27f726a3db2faafa9214b3a4ee3d" + }, + "checkpoints/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-128/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-128/adapter_model.safetensors": { + "size": 547777976, + "sha256": "12f230530c54d3f2f44d3f6846d2a5c55bdb6475036bdc98968be3c5c665a155" + }, + "checkpoints/checkpoint-128/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-128/optimizer.pt": { + "size": 1048106435, + "sha256": "0a46a2e5ceaf9e4f3f7eaf5888b0030e383d8f02c8c601cfdf3bfca1042eb938" + }, + "checkpoints/checkpoint-128/rng_state.pth": { + "size": 14645, + "sha256": "92697a8e0f68025132ee5d9834b875806811d123b7dc25857d9b657ce029e1ca" + }, + "checkpoints/checkpoint-128/scheduler.pt": { + "size": 1465, + "sha256": "efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f" + }, + "checkpoints/checkpoint-128/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-128/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-128/tokens_state.json": { + "size": 38, + "sha256": "4269445b0c90b323256b711674701ba494774adc0ad359339f0acb0914a955af" + }, + "checkpoints/checkpoint-128/trainer_state.json": { + "size": 56168, + "sha256": "d8ff9c17de4192917b5bb1913b7663ebb0f8af729b12cf4f188194c94e89bafb" + }, + "checkpoints/checkpoint-128/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-160/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-160/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-160/adapter_model.safetensors": { + "size": 547777976, + "sha256": "cb09b512ed97e9c1471f4bf12ff053743a9ae143b3eb48010e034fb0e01b2206" + }, + "checkpoints/checkpoint-160/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-160/optimizer.pt": { + "size": 1048106435, + "sha256": "53e9ba68f36679e6cfe39be8a9c321dc9071aa233fc3654a1f7aaf1ea9e50499" + }, + "checkpoints/checkpoint-160/rng_state.pth": { + "size": 14645, + "sha256": "d3d8d99873a308b8540f12cd8c43a40b60230c9c2eccf247652e1b5b269679cd" + }, + "checkpoints/checkpoint-160/scheduler.pt": { + "size": 1465, + "sha256": "69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89" + }, + "checkpoints/checkpoint-160/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-160/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-160/tokens_state.json": { + "size": 38, + "sha256": "51c7480cb8714b1956f770d400c9e974870f7b09a6a65e9fc38d47997df68cc7" + }, + "checkpoints/checkpoint-160/trainer_state.json": { + "size": 70144, + "sha256": "da9160a61262545ab7048098f24b23d475a44fe860e3f3134973e647a37d5d06" + }, + "checkpoints/checkpoint-160/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-192/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-192/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-192/adapter_model.safetensors": { + "size": 547777976, + "sha256": "e11c0436419262a8d7dffabe6ab36e81648ca4d1a983e89a6d8cd144e5e75faf" + }, + "checkpoints/checkpoint-192/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-192/optimizer.pt": { + "size": 1048106435, + "sha256": "43f0be6f7706c4a9910a7d43b27df21ebe3e39389dd5b7007be5ed414249cd51" + }, + "checkpoints/checkpoint-192/rng_state.pth": { + "size": 14645, + "sha256": "3e7d3b55fe6373e1452968ec327f795eb851f8990c862c581f02fa825e7c7f48" + }, + "checkpoints/checkpoint-192/scheduler.pt": { + "size": 1465, + "sha256": "076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd" + }, + "checkpoints/checkpoint-192/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-192/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-192/tokens_state.json": { + "size": 38, + "sha256": "035ad54fc8bbaa07ee17c6a4371faa6eea636f2fef98a7ed8b3290a86bccdbbd" + }, + "checkpoints/checkpoint-192/trainer_state.json": { + "size": 84131, + "sha256": "c22f182b9816733e88cbca227c6277a47e6b942ee185ef40f5566829930c8965" + }, + "checkpoints/checkpoint-192/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-224/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-224/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-224/adapter_model.safetensors": { + "size": 547777976, + "sha256": "9f72f8aca7074d1b11068a5a4c4b98c07550a8a7b8ed04ef1ddcdf6d0a4d0a9a" + }, + "checkpoints/checkpoint-224/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-224/optimizer.pt": { + "size": 1048106435, + "sha256": "9c8e58da8905564001a01dd91d692e9e3860e966e43efc8831b6d7748dcd55fa" + }, + "checkpoints/checkpoint-224/rng_state.pth": { + "size": 14645, + "sha256": "abf6a2d5c2e9ead13ead5ac022f80d62ef7845bff3464b128b8221bee281b403" + }, + "checkpoints/checkpoint-224/scheduler.pt": { + "size": 1465, + "sha256": "f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822" + }, + "checkpoints/checkpoint-224/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-224/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-224/tokens_state.json": { + "size": 39, + "sha256": "b4410072e6d2f038bdcbc9428dd9988b1e103722fa53446ba0fafb06e31b689d" + }, + "checkpoints/checkpoint-224/trainer_state.json": { + "size": 98140, + "sha256": "7c6e950f726cd3200774074bcf82258632e738240e08d6df1f4cb7aadbc760c7" + }, + "checkpoints/checkpoint-224/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-256/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-256/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-256/adapter_model.safetensors": { + "size": 547777976, + "sha256": "d742113c99707c7c9623f4899c9f29701c09d638baeca2dbe1057c9868c078b3" + }, + "checkpoints/checkpoint-256/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-256/optimizer.pt": { + "size": 1048106435, + "sha256": "a1dd8411289f9e180f11bde00ec6713bdd67229f322c9600a2e668c543f59e69" + }, + "checkpoints/checkpoint-256/rng_state.pth": { + "size": 14645, + "sha256": "ceacb5e4edaa8698c0ee97119b94398f0f0aea14574bfd7c59026540a8cf3390" + }, + "checkpoints/checkpoint-256/scheduler.pt": { + "size": 1465, + "sha256": "1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3" + }, + "checkpoints/checkpoint-256/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-256/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-256/tokens_state.json": { + "size": 39, + "sha256": "85cd4c82422c69da428e8b614ca86d93b8f986debe2250175a9c36d250d6c37f" + }, + "checkpoints/checkpoint-256/trainer_state.json": { + "size": 112176, + "sha256": "3ab019d25c41ed4c9df14da67922ba645038d394b691929496257347702305f4" + }, + "checkpoints/checkpoint-256/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-288/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-288/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-288/adapter_model.safetensors": { + "size": 547777976, + "sha256": "7e3717759c8739a3463e6856b72c914a12658b6c60fb58ed8c6c7ca716aedc4b" + }, + "checkpoints/checkpoint-288/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-288/optimizer.pt": { + "size": 1048106435, + "sha256": "f3b1c0ccf8e5ee0ced1adcb642f8399722cd319e3f12c89703e6b1ae55c8e3bd" + }, + "checkpoints/checkpoint-288/rng_state.pth": { + "size": 14645, + "sha256": "7fbae4a4d555f312b39a39b46b1af77fb8085c6d4427f0ca32297157fb1ece4f" + }, + "checkpoints/checkpoint-288/scheduler.pt": { + "size": 1465, + "sha256": "a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363" + }, + "checkpoints/checkpoint-288/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-288/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-288/tokens_state.json": { + "size": 39, + "sha256": "a989ecde5ae2cd6c8cf8dd0819a235d6067b5410f339d89e35bd0adc47a50a06" + }, + "checkpoints/checkpoint-288/trainer_state.json": { + "size": 126236, + "sha256": "d1be3634a8fee2db65b822695f9268d5c1a323ea48d09612846d7e66cc74a70f" + }, + "checkpoints/checkpoint-288/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-32/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-32/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-32/adapter_model.safetensors": { + "size": 547777976, + "sha256": "65860e6f7cf9447e039d8e46c6ea335960033027c510fff5f17c38338bb234a6" + }, + "checkpoints/checkpoint-32/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-32/optimizer.pt": { + "size": 1048106435, + "sha256": "ec41e716968f92f8039a3be34747f099ce1994d33eb2c99a67ae32bb677c8cc0" + }, + "checkpoints/checkpoint-32/rng_state.pth": { + "size": 14645, + "sha256": "046455cb6ded9b2989f35bf5efbbff0517d6651771b52c93d31759906cafe479" + }, + "checkpoints/checkpoint-32/scheduler.pt": { + "size": 1465, + "sha256": "8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc" + }, + "checkpoints/checkpoint-32/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-32/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-32/tokens_state.json": { + "size": 37, + "sha256": "604697e527533defbb346641f0b353ca7ae9ef38695a9435ecfcec09015c38de" + }, + "checkpoints/checkpoint-32/trainer_state.json": { + "size": 14410, + "sha256": "829fdc8317ca1b9a9df48fa426f80bc3c9f1d564414d2df532ec231d50047c4d" + }, + "checkpoints/checkpoint-32/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-320/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-320/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-320/adapter_model.safetensors": { + "size": 547777976, + "sha256": "6d0af4d5d51bd8ca39574aaed0877128d4353fde72d015ff50c83d68c340a811" + }, + "checkpoints/checkpoint-320/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-320/optimizer.pt": { + "size": 1048106435, + "sha256": "1e91bd2c23cccc34f21d3820890a1bc3306d5b2d1bf7834462936463ae973261" + }, + "checkpoints/checkpoint-320/rng_state.pth": { + "size": 14645, + "sha256": "b38f18d0c656cdb6b3a38bf169f91bd05f72a5d1b092c121195d69fe47abe823" + }, + "checkpoints/checkpoint-320/scheduler.pt": { + "size": 1465, + "sha256": "e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0" + }, + "checkpoints/checkpoint-320/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-320/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-320/tokens_state.json": { + "size": 39, + "sha256": "c4734afd6b91080c4d1b45b0c604d3ac5ac748c61c08f5cc86a91e0e1c48391d" + }, + "checkpoints/checkpoint-320/trainer_state.json": { + "size": 140300, + "sha256": "7eab294a4cafd21abd317d6fa292bc1c723494b182c278412118fd0de76269ab" + }, + "checkpoints/checkpoint-320/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-352/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-352/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-352/adapter_model.safetensors": { + "size": 547777976, + "sha256": "0575458d5c9a04122cf7f38cb432153118e3c843bde848cc878ecd5d40e8b7d5" + }, + "checkpoints/checkpoint-352/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-352/optimizer.pt": { + "size": 1048106435, + "sha256": "501397c9ff895c18412afa96acda0dbd4547fa44776f0d057547a72cdf51dc2e" + }, + "checkpoints/checkpoint-352/rng_state.pth": { + "size": 14645, + "sha256": "4e43b4a1e67b4766319419b948066927c1e998670ed9c2c2d0ce488de319be4d" + }, + "checkpoints/checkpoint-352/scheduler.pt": { + "size": 1465, + "sha256": "153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7" + }, + "checkpoints/checkpoint-352/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-352/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-352/tokens_state.json": { + "size": 40, + "sha256": "09843bc625dabff7e312476736aa4046945aa6251df4f2520f850b2888c923b1" + }, + "checkpoints/checkpoint-352/trainer_state.json": { + "size": 154409, + "sha256": "a43a49e697fc6ac3e0cd214b11fcfb590bbeddac02f89de3926fb70fc558f904" + }, + "checkpoints/checkpoint-352/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-384/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-384/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-384/adapter_model.safetensors": { + "size": 547777976, + "sha256": "6e12e316c4b40808347cd4df07ca3a1fc4190596cafc5155d40ea2a381b9cbda" + }, + "checkpoints/checkpoint-384/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-384/optimizer.pt": { + "size": 1048106435, + "sha256": "87b2da364ff6d27ddfaa2bc1be957c48a94448d01c409a4a53226d6be598ede6" + }, + "checkpoints/checkpoint-384/rng_state.pth": { + "size": 14645, + "sha256": "f0f5fee0fb94cee084227608506532dac3dd7e8758af76be7848e1b7809a65f2" + }, + "checkpoints/checkpoint-384/scheduler.pt": { + "size": 1465, + "sha256": "8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79" + }, + "checkpoints/checkpoint-384/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-384/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-384/tokens_state.json": { + "size": 40, + "sha256": "01b503867d135f43ce4319a512665807cec9cd559ff77c805ba2a6ff3ece3c88" + }, + "checkpoints/checkpoint-384/trainer_state.json": { + "size": 168547, + "sha256": "542e1cc49e45abc06d18570bf459feef7eb4e4536b960511e2f15619c3cd1932" + }, + "checkpoints/checkpoint-384/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-416/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-416/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-416/adapter_model.safetensors": { + "size": 547777976, + "sha256": "17b9b01936d3c08a85b19bc450a46a3c542df7b4ccdfbce4eb246466623e95b6" + }, + "checkpoints/checkpoint-416/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-416/optimizer.pt": { + "size": 1048106435, + "sha256": "fd0d036472778760752be6def787b8de9dc1d5e2ec26c9c978ea57952b6712dd" + }, + "checkpoints/checkpoint-416/rng_state.pth": { + "size": 14645, + "sha256": "64b177cec70f807704a17979be33619030d753ef926c4ec9201771f63712df24" + }, + "checkpoints/checkpoint-416/scheduler.pt": { + "size": 1465, + "sha256": "2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc" + }, + "checkpoints/checkpoint-416/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-416/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-416/tokens_state.json": { + "size": 40, + "sha256": "2683426d06441b84580eb4527f757b667df15e13127b065d0abf4b7ef281d953" + }, + "checkpoints/checkpoint-416/trainer_state.json": { + "size": 182660, + "sha256": "4e47313c926f14aec74078b59f0b64eb75647f74187b17b3470f955843d33056" + }, + "checkpoints/checkpoint-416/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-448/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-448/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-448/adapter_model.safetensors": { + "size": 547777976, + "sha256": "f0b323f3fea45e02c2fabd61c429c6bdc140e7b3425cdbb0bc409168a0309a1f" + }, + "checkpoints/checkpoint-448/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-448/optimizer.pt": { + "size": 1048106435, + "sha256": "5d60414e59da811244f04cd863580479c781287bf8ed3e16305b69047c61fb7f" + }, + "checkpoints/checkpoint-448/rng_state.pth": { + "size": 14645, + "sha256": "3aa9afe585253001856afc47614f0a2780fb092e3570494e6b46234b23895a17" + }, + "checkpoints/checkpoint-448/scheduler.pt": { + "size": 1465, + "sha256": "457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503" + }, + "checkpoints/checkpoint-448/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-448/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-448/tokens_state.json": { + "size": 40, + "sha256": "8fa03c74384c1e8dbb7b43f984e59232d97211cb7670822b469abad26a1331be" + }, + "checkpoints/checkpoint-448/trainer_state.json": { + "size": 196792, + "sha256": "04eeb8bf1419269c866e49230c724b9f88490d3ec66e522bee5bdf489820cfec" + }, + "checkpoints/checkpoint-448/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-480/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-480/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-480/adapter_model.safetensors": { + "size": 547777976, + "sha256": "e767f38d93516d2a91d9db9148bc7a639c69596a4261e7cf09dc557e23edead2" + }, + "checkpoints/checkpoint-480/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-480/optimizer.pt": { + "size": 1048106435, + "sha256": "598167917e6ecb20f95ae116c75b2a2dbff7f53e6763d0c4a52cea2e1b44bc94" + }, + "checkpoints/checkpoint-480/rng_state.pth": { + "size": 14645, + "sha256": "7e112da89198d46f80dee18f0562406093076519aaf838b7006fe23784d97abc" + }, + "checkpoints/checkpoint-480/scheduler.pt": { + "size": 1465, + "sha256": "1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01" + }, + "checkpoints/checkpoint-480/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-480/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-480/tokens_state.json": { + "size": 40, + "sha256": "0a1628399182a1c24291522bcc09ced9cce8d3227ffba2d7fa81c29ab33245ab" + }, + "checkpoints/checkpoint-480/trainer_state.json": { + "size": 210928, + "sha256": "f44911cb2677808a994de1220f943a7cbfa3eb0f465c152bd0f9b26a5d975b4f" + }, + "checkpoints/checkpoint-480/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-512/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-512/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-512/adapter_model.safetensors": { + "size": 547777976, + "sha256": "6f7da5abb2cf5cc22da25374898331b609fa27f726a3db2faafa9214b3a4ee3d" + }, + "checkpoints/checkpoint-512/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-512/optimizer.pt": { + "size": 1048106435, + "sha256": "e4af1671343e680b761c384855b03b2f3edba0c1c73e238f273b4ec981203dd8" + }, + "checkpoints/checkpoint-512/rng_state.pth": { + "size": 14645, + "sha256": "ad796432b92ca02dee812ca2f0084633293af1832f48212de91b45c8633bc0c9" + }, + "checkpoints/checkpoint-512/scheduler.pt": { + "size": 1465, + "sha256": "697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4" + }, + "checkpoints/checkpoint-512/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-512/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-512/tokens_state.json": { + "size": 40, + "sha256": "8c9597e2c7dbc41758b5666f38c7be3c9967ffdee92b9e99bc35a6547845b118" + }, + "checkpoints/checkpoint-512/trainer_state.json": { + "size": 225098, + "sha256": "b90b21d6af9eabe402ffa0ee283ba4136d11677977d27ca92576ff46dec396a1" + }, + "checkpoints/checkpoint-512/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-64/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-64/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-64/adapter_model.safetensors": { + "size": 547777976, + "sha256": "405f4b9c343f437b8c4364d7212cae510d9d8adbf2c91918e11fbd8eaa766321" + }, + "checkpoints/checkpoint-64/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-64/optimizer.pt": { + "size": 1048106435, + "sha256": "e57e479684c5374f4a3e01aa49e216786c3b13dffcf2843772779a92d66d5d4b" + }, + "checkpoints/checkpoint-64/rng_state.pth": { + "size": 14645, + "sha256": "0c1c39dd1f41c6a2efef09bbfbb45aea5f6c893008225984765de40d9ce68a16" + }, + "checkpoints/checkpoint-64/scheduler.pt": { + "size": 1465, + "sha256": "15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503" + }, + "checkpoints/checkpoint-64/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-64/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-64/tokens_state.json": { + "size": 38, + "sha256": "1d99cc06f1f29116e45c0b8b9f4e6c7e27d1bbb9c68f6fbe7e8c968deb5aa5ad" + }, + "checkpoints/checkpoint-64/trainer_state.json": { + "size": 28314, + "sha256": "00035d1f0b3680e6736f94d8a95bc5ab1ec56215adc1c7eba659615a8e4b4668" + }, + "checkpoints/checkpoint-64/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/checkpoint-96/README.md": { + "size": 5208, + "sha256": "c84bbe0f2c0fc58818de1d0283fa9d8436340bbb63447c6e6b7358a8320be210" + }, + "checkpoints/checkpoint-96/adapter_config.json": { + "size": 1098, + "sha256": "d7b7faff72e4b5c3b6707346ec6c048dcec9ae30dae4ce8a710d0945027d0370" + }, + "checkpoints/checkpoint-96/adapter_model.safetensors": { + "size": 547777976, + "sha256": "ee64598865679ab0749385b9a85d0fa758de86dfe72510c7116064aef45ddede" + }, + "checkpoints/checkpoint-96/chat_template.jinja": { + "size": 1532, + "sha256": "7de1c58e208eda46e9c7f86397df37ec49883aeece39fb961e0a6b24088dd3c4" + }, + "checkpoints/checkpoint-96/optimizer.pt": { + "size": 1048106435, + "sha256": "2e0ff534f3d7cfd3ac1b559bfc3cd0ec2b5785000abac241e533c30e495f0721" + }, + "checkpoints/checkpoint-96/rng_state.pth": { + "size": 14645, + "sha256": "adbeb73951e9ff069c0226d3d5550f1fe22c83b0c416ebc3684c1f8fff3f48ce" + }, + "checkpoints/checkpoint-96/scheduler.pt": { + "size": 1465, + "sha256": "786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0" + }, + "checkpoints/checkpoint-96/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/checkpoint-96/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints/checkpoint-96/tokens_state.json": { + "size": 38, + "sha256": "9cb16f72fca12453d18ed81180856b46900e3ea3f65c2d9869741fc52eb2722e" + }, + "checkpoints/checkpoint-96/trainer_state.json": { + "size": 42241, + "sha256": "c71cea16f5a2ba7a9879ad7ff01cf1886ca9bfb5ddcf34c0c503896165b12316" + }, + "checkpoints/checkpoint-96/training_args.bin": { + "size": 8273, + "sha256": "406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b" + }, + "checkpoints/config.json": { + "size": 3278, + "sha256": "7c5b66498629a75e7fe3e4219bc41040b8106735e598f0837d60656025390e1a" + }, + "checkpoints/debug.log": { + "size": 267479, + "sha256": "0c1cf59963ce640b906e3df9d3d27da317ca2df067c36bf061281a66c7217d3b" + }, + "checkpoints/processor_config.json": { + "size": 519, + "sha256": "e58dda857eb60dae48a0146bedd13f9e4664f4066d6269f1eaa934db8f2d704f" + }, + "checkpoints/tokenizer.json": { + "size": 33384567, + "sha256": "daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726" + }, + "checkpoints/tokenizer_config.json": { + "size": 746, + "sha256": "985e3a86edf82aa70f471b4f974e2fea2c6bac9bfb57edd073d0679f6ecaaae6" + }, + "checkpoints.jsonl": { + "size": 141, + "sha256": "f9ebd414bb02ac8239ec8648ecf483002b2afe26dfc138486e69202fa52337a6" + }, + "ckpt_control_matched__charter2.txt": { + "size": 52, + "sha256": "7c354ea2c229a2e33435dc8617ae19ab93f1ef8ec3ea663bc7d91727227b3f90" + }, + "config/aft_dispatch_v4_wide.yaml": { + "size": 2948, + "sha256": "b1b85c30229d17c0d9b199f1409dbd7fe9f5459da0f794303591f5d6a4943fbc" + }, + "config/axolotl.yaml": { + "size": 1210, + "sha256": "85e1ac7519356eb24741e70e76c15262c684b41306bccddca8d2bf7f8925e98b" + }, + "health/training_started.json": { + "size": 137, + "sha256": "1fe9af65ede9bff150b1f6ab709f9197ce5492b76adff1ddd561b16cb7f6e0f9" + }, + "run.json": { + "size": 409, + "sha256": "4218d4f7092563b3380098e87cb9c44391c180eb2390c17158cf0da92d8c9b78" + }, + "train.log": { + "size": 274871, + "sha256": "5fbfa6d885701f1dd217d2687f12a05016ee15562e59d83325c6861d99c95b5f" + }, + "trainer_state.final.json": { + "size": 225098, + "sha256": "b90b21d6af9eabe402ffa0ee283ba4136d11677977d27ca92576ff46dec396a1" + }, + "training_examples.jsonl": { + "size": 3165023, + "sha256": "152527073625fe282aef5507a557d25d9668ced86e5274c56f2f8df319ffde3a" + }, + "training_provenance.json": { + "size": 4088, + "sha256": "0441d17f6c6b3e1130599f70918db9bab311456d687b3d541eede77bb633a014" + }, + "training_trace.jsonl": { + "size": 181830, + "sha256": "5ae9db059a214bedaacf7c689e525cb8ace877c4046c2f14eb6fef11672d3194" + } + } +} diff --git a/aft_wave_v2/control_matched__charter2/training/TRAINED.json b/aft_wave_v2/control_matched__charter2/training/TRAINED.json new file mode 100644 index 0000000000000000000000000000000000000000..6132df38be0bbcaefc8408db3f96d89521c4bfba --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/TRAINED.json @@ -0,0 +1,55 @@ +{ + "version": "dispatch_wave_v2", + "arm": "control_matched__charter2", + "parameterization": "lora", + "parent_repo": "arcadia-impact/scimt-dispatch-models", + "parent_prefix": "gate2_midtrain4/dolmino/post_dolci100", + "dataset_sha256": "6a1f783d80a3d91f3aea0f9e8fb701f65f1e60162ac5be0f24db62b3c7612b57", + "training_rows": 8192, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "minutes": 57.94, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + }, + "optimizer_steps": 512, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "eval_steps": [ + 32, + 64, + 128, + 256, + 512 + ], + "optimizer_state_saved": true +} diff --git a/aft_wave_v2/control_matched__charter2/training/axolotl.yaml b/aft_wave_v2/control_matched__charter2/training/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f07d72e90f5f51fa00b4f36166cfef8885f4ad8a --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/axolotl.yaml @@ -0,0 +1,56 @@ +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_charter2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoint.json b/aft_wave_v2/control_matched__charter2/training/checkpoint.json new file mode 100644 index 0000000000000000000000000000000000000000..fa428f9831c5b33357c7a5ae8277c753a0b4f0e7 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoint.json @@ -0,0 +1,74 @@ +{ + "experiment": "scimt-train:control_matched__charter2", + "spec": null, + "kind": null, + "note": "Spec-free training stage (scimt.train.train_dataset) \u2014 a post-training link in a staged chain, not a spec install.", + "train": { + "data": "/workspace/wave/data/datasets/aft_charter2.jsonl", + "dataset_meta": { + "adhoc": true + }, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "load_checkpoint_path": "/workspace/wave/parent", + "grpo": null, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + } + }, + "run_name": "control_matched__charter2", + "pointer_file": "/workspace/wave/training/ckpt_control_matched__charter2.txt", + "backend": "axolotl", + "sampler": "/workspace/wave/training/checkpoints/checkpoint-512", + "state": "/workspace/wave/training/checkpoints/checkpoint-512", + "model": "gemma3_12b_it", + "meta": { + "experiment": "scimt-train:control_matched__charter2", + "spec": null, + "kind": null, + "note": "Spec-free training stage (scimt.train.train_dataset) \u2014 a post-training link in a staged chain, not a spec install.", + "train": { + "data": "/workspace/wave/data/datasets/aft_charter2.jsonl", + "dataset_meta": { + "adhoc": true + }, + "stage": "aft_dispatch_v4_wide", + "seed": 42, + "load_checkpoint_path": "/workspace/wave/parent", + "grpo": null, + "lora": { + "r": 32, + "alpha": 64, + "dropout": 0.05, + "target_linear": false, + "target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "initial_adapter_path": null + } + }, + "run_name": "control_matched__charter2", + "pointer_file": "/workspace/wave/training/ckpt_control_matched__charter2.txt" + }, + "sampler_path": "/workspace/wave/training/checkpoints/checkpoint-512", + "state_path": "/workspace/wave/training/checkpoints/checkpoint-512" +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints.jsonl b/aft_wave_v2/control_matched__charter2/training/checkpoints.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..0a5fa00d7287f6920786ad43da67f36b8568e07a --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints.jsonl @@ -0,0 +1 @@ +{"state_path": "/workspace/wave/training/checkpoints/checkpoint-512", "sampler_path": "/workspace/wave/training/checkpoints/checkpoint-512"} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/README.md new file mode 100644 index 0000000000000000000000000000000000000000..5d0ae0247b7ddbbdc0d7896f27d6b060f1f13d99 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/README.md @@ -0,0 +1,128 @@ +--- +library_name: peft +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +datasets: +- /workspace/wave/data/datasets/aft_charter2.jsonl +base_model: /workspace/wave/parent +pipeline_tag: text-generation +model-index: +- name: workspace/wave/training/checkpoints + results: [] +--- + + + +[Built with Axolotl](https://github.com/axolotl-ai-cloud/axolotl) +
See axolotl config + +axolotl version: `0.17.0` +```yaml +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_charter2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj + +``` + +

+ +# workspace/wave/training/checkpoints + +This model was trained from scratch on the /workspace/wave/data/datasets/aft_charter2.jsonl dataset. + +## Model description + +More information needed + +## Intended uses & limitations + +More information needed + +## Training and evaluation data + +More information needed + +## Training procedure + +### Training hyperparameters + +The following hyperparameters were used during training: +- learning_rate: 0.0001 +- train_batch_size: 16 +- eval_batch_size: 16 +- seed: 42 +- gradient_accumulation_steps: 2 +- total_train_batch_size: 32 +- optimizer: Use OptimizerNames.ADAMW_TORCH_FUSED with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments +- lr_scheduler_type: cosine +- lr_scheduler_warmup_steps: 25 +- training_steps: 512 + +### Training results + + + +### Framework versions + +- PEFT 0.19.1 +- Transformers 5.9.0 +- Pytorch 2.12.1+cu126 +- Datasets 4.8.5 +- Tokenizers 0.22.2 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5bc2f0988df480edd572b506b460277309280203 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6f7da5abb2cf5cc22da25374898331b609fa27f726a3db2faafa9214b3a4ee3d +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..f14d4a8262bf901eca4109c4190c162cf303a09e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:12f230530c54d3f2f44d3f6846d2a5c55bdb6475036bdc98968be3c5c665a155 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..2c0a204a7ddf31e2a13e93a2fe3c628c8761c61f --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0a46a2e5ceaf9e4f3f7eaf5888b0030e383d8f02c8c601cfdf3bfca1042eb938 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..16431a1bf62d496d323a720f2d3d85d1568643e6 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92697a8e0f68025132ee5d9834b875806811d123b7dc25857d9b657ce029e1ca +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..bc9eb8d587c7c4d2a9e80caed8c0953081dce4ba --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:efd85ddb91fbafff2e34e19b252134ec33fb00857c2936b417257fd723d5c25f +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f4d979b038724d80bdb7fbb662e65dabd7361d44 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/tokens_state.json @@ -0,0 +1 @@ +{"total": 3872816, "trainable": 58497} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0cd6911f8d45622756d044e56ad7c7bbd8d84449 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/trainer_state.json @@ -0,0 +1,1826 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.5, + "eval_steps": 500, + "global_step": 128, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2.628706536652631e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-128/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..834fbed420006140d93e83081d52b6e79c9969da --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cb09b512ed97e9c1471f4bf12ff053743a9ae143b3eb48010e034fb0e01b2206 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..dc659b7cea4c66648b03b15273de7db16bffe0f5 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:53e9ba68f36679e6cfe39be8a9c321dc9071aa233fc3654a1f7aaf1ea9e50499 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..fcf764f71de8ec0f7c86dde47df0b3e12fe5d3c2 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d3d8d99873a308b8540f12cd8c43a40b60230c9c2eccf247652e1b5b269679cd +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..d47c908eab228a9c923b0bfb0a7fb1a6dfa773e6 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:69d150a62f01954562efb0e1e0f0cfe1a6c1a63dcb6f19a50e14230cf65a7b89 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..cf898473d445e48734a4be2b6f1e8f3cdbff88e0 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/tokens_state.json @@ -0,0 +1 @@ +{"total": 4836688, "trainable": 73119} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7125be4ed1de4687a3136721068db4c5f81db35b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/trainer_state.json @@ -0,0 +1,2274 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.625, + "eval_steps": 500, + "global_step": 160, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.282942789264799e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-160/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..59f8ddec904fa3164b2257f2a30e5fe092e70a2c --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e11c0436419262a8d7dffabe6ab36e81648ca4d1a983e89a6d8cd144e5e75faf +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b5fe0f38cd74bb38df78ab64147b8a301aef9bf8 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:43f0be6f7706c4a9910a7d43b27df21ebe3e39389dd5b7007be5ed414249cd51 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..6a1c2715da5b33e568cd2c8254c3db8d081c58df --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3e7d3b55fe6373e1452968ec327f795eb851f8990c862c581f02fa825e7c7f48 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..68fafd70d4f03fc26bd72aefbdcd22105c983415 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:076826708eb68e7e0324ae251ebe1ef8facf8800fbc47028be05da96491371fd +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..293c2037757a64a76e8d4c4183e4f84df6f99631 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/tokens_state.json @@ -0,0 +1 @@ +{"total": 5807952, "trainable": 87814} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..333faebc8ebc81654ad6383f9d276d3dd2de58cc --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/trainer_state.json @@ -0,0 +1,2722 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.75, + "eval_steps": 500, + "global_step": 192, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 3.942196424246523e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-192/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..72a9bab1f44531edf2cb460d0ee33dba60ea6a6f --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9f72f8aca7074d1b11068a5a4c4b98c07550a8a7b8ed04ef1ddcdf6d0a4d0a9a +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..11ac3c78d48f97f37513d4a04c39b901d9007963 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9c8e58da8905564001a01dd91d692e9e3860e966e43efc8831b6d7748dcd55fa +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bb931b8cb4bb21cfdf7f3f2b3f40fa27bdf0b107 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:abf6a2d5c2e9ead13ead5ac022f80d62ef7845bff3464b128b8221bee281b403 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..734e9a25165547994d3add43bbfb2dd65b7e559d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f01c79694697cdd875b0827789740f4369175c3cc86ff2137cdbb9f8780c0822 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..88a1b3b3658254a1760191b16cebc87155e42e44 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/tokens_state.json @@ -0,0 +1 @@ +{"total": 6774432, "trainable": 102463} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..dd28af4c606449f22c71c572d576a8e81391e1d6 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/trainer_state.json @@ -0,0 +1,3170 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.875, + "eval_steps": 500, + "global_step": 224, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4.598202878863534e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-224/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..bc479483b834c7010bd81a486690c4fb566b14fa --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d742113c99707c7c9623f4899c9f29701c09d638baeca2dbe1057c9868c078b3 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..9bc099761dd57d58015506990b9c31414aca3cf1 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a1dd8411289f9e180f11bde00ec6713bdd67229f322c9600a2e668c543f59e69 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..56111c71650250afe5982772370c501f364c36b9 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ceacb5e4edaa8698c0ee97119b94398f0f0aea14574bfd7c59026540a8cf3390 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..0d402b407a8fd89f8bcb5de6f509c9979553d18a --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1078fafd95411b83b445384e23d0fd62bdb653339321029eb68f5bd5168093f3 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..98801750b4bd2bee0f4368f9504bdf4561a04fe2 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/tokens_state.json @@ -0,0 +1 @@ +{"total": 7741936, "trainable": 117062} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..da6ee98dbed049c65fca7e7c67d204ee531d1ef0 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/trainer_state.json @@ -0,0 +1,3618 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 256, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.254904382120484e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-256/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..234780b3c6e4b08d53d2d628564d45a8e914c938 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e3717759c8739a3463e6856b72c914a12658b6c60fb58ed8c6c7ca716aedc4b +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..6da161850a32f5f472120ef496d467d4ff70fbbc --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f3b1c0ccf8e5ee0ced1adcb642f8399722cd319e3f12c89703e6b1ae55c8e3bd +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4958ce4847b78dd3cf626a5e3bfadf2bda0cc0eb --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7fbae4a4d555f312b39a39b46b1af77fb8085c6d4427f0ca32297157fb1ece4f +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..5f00d58c65c626a11e09272704f772fedb532412 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a3b676d4693f7202d0a8375b6101005aaf04d716d840a44d0b3c3ec839a7b363 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..50c07a686ae9c20448a94b99f49a860983a96eb1 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/tokens_state.json @@ -0,0 +1 @@ +{"total": 8713056, "trainable": 131943} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..768490b7631f578ce1520c07aa67117ed74f358d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/trainer_state.json @@ -0,0 +1,4066 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.125, + "eval_steps": 500, + "global_step": 288, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5.914060275887217e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-288/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..33a85363478f6df3132714fdb7b6caa07beaf076 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:65860e6f7cf9447e039d8e46c6ea335960033027c510fff5f17c38338bb234a6 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..928f17942d45bb6be10e4c9f98e57b40de9a970e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ec41e716968f92f8039a3be34747f099ce1994d33eb2c99a67ae32bb677c8cc0 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..bcaad7482cc1813859d1e652f169fa68f16981e3 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:046455cb6ded9b2989f35bf5efbbff0517d6651771b52c93d31759906cafe479 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..02f9ed8e0defc2ebe0d43c0d14abc27c32e2b2cf --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c4c9564eeed32d66a97d93c881b32c0e6dbd47c4a8382e1a27d79ac3aa7fefc +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..772b88c488dba60cfffa940e388aaa1062089c18 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/tokens_state.json @@ -0,0 +1 @@ +{"total": 969264, "trainable": 14680} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..0ab407fd407beb8dff4df8249681a0ed9fc22e93 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/trainer_state.json @@ -0,0 +1,482 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.125, + "eval_steps": 500, + "global_step": 32, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.578961181068442e+16, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-32/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..3aa2232fbea282f084cbff81f6cb7385bf352bfb --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6d0af4d5d51bd8ca39574aaed0877128d4353fde72d015ff50c83d68c340a811 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..fcc12035aca2f6b73ed1870b6fd0c2914aef60c6 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1e91bd2c23cccc34f21d3820890a1bc3306d5b2d1bf7834462936463ae973261 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4f3aa8844b4edc457bc8f75976c2cd9477264213 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b38f18d0c656cdb6b3a38bf169f91bd05f72a5d1b092c121195d69fe47abe823 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..cf4669ac28031073c15cff33e6a2bd7daa0e7337 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e084f01555147a3a8176886c943886b324735c8a293c0544952d61dda60efdb0 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2fa90e77e8c7e05edefd1584b5d553e45bb2a6ed --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/tokens_state.json @@ -0,0 +1 @@ +{"total": 9682672, "trainable": 146693} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4480f29033fd3f5f2487b38f58802ce6bd6ac2ae --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/trainer_state.json @@ -0,0 +1,4514 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.25, + "eval_steps": 500, + "global_step": 320, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 6.57219531696404e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-320/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..0ebeefe07144b68fb41c3d0b132b914470ec8f71 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0575458d5c9a04122cf7f38cb432153118e3c843bde848cc878ecd5d40e8b7d5 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..b65bebcc2e3be59f8ba66c969d81d2fb0ff4ab77 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:501397c9ff895c18412afa96acda0dbd4547fa44776f0d057547a72cdf51dc2e +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..edfdf773a7ab6e4f74110dc00b2762630494db6f --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4e43b4a1e67b4766319419b948066927c1e998670ed9c2c2d0ce488de319be4d +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..1d2b8d34c008174b230d6dbbee4469d934686b9f --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:153174d2675ff0dc957d8edec3db026478f6ffc8ae455dadcc1c6ec96b4b4ce7 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..3b422e5fcdb4151b89b39a69f1a5978bb874294e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/tokens_state.json @@ -0,0 +1 @@ +{"total": 10654032, "trainable": 161132} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..9a9aa39500ce00491b8f1dd010a2e18e39e318f0 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/trainer_state.json @@ -0,0 +1,4962 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.375, + "eval_steps": 500, + "global_step": 352, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.062361083924770355, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.0006638169870711863, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 321, + "tokens/total": 9712832, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 147139 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.009022507816553116, + "learning_rate": 4.004940255778431e-05, + "loss": 0.00017464160919189453, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00017, + "step": 322, + "tokens/total": 9743344, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 147589 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.38307738304138184, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0034184439573436975, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00342, + "step": 323, + "tokens/total": 9773760, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 148045 + }, + { + "epoch": 1.265625, + "grad_norm": 0.3203769028186798, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0051149846985936165, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00513, + "step": 324, + "tokens/total": 9804080, + "tokens/train_per_sec_per_gpu": 37.47, + "tokens/trainable": 148494 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.7488558292388916, + "learning_rate": 3.923084939213296e-05, + "loss": 0.01595677062869072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01608, + "step": 325, + "tokens/total": 9834400, + "tokens/train_per_sec_per_gpu": 36.84, + "tokens/trainable": 148957 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.009091824293136597, + "learning_rate": 3.895929566172861e-05, + "loss": 0.00014956737868487835, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 326, + "tokens/total": 9864704, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 149451 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.005110942758619785, + "learning_rate": 3.868840945050728e-05, + "loss": 0.00013913170550949872, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00014, + "step": 327, + "tokens/total": 9894656, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 149904 + }, + { + "epoch": 1.28125, + "grad_norm": 0.007466079201549292, + "learning_rate": 3.841820203114995e-05, + "loss": 0.00013769841461908072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00014, + "step": 328, + "tokens/total": 9924976, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 150378 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.16560006141662598, + "learning_rate": 3.814868464809027e-05, + "loss": 0.009101686999201775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00914, + "step": 329, + "tokens/total": 9955376, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 150853 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.01060777809470892, + "learning_rate": 3.787986851704667e-05, + "loss": 0.00019600678933784366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0002, + "step": 330, + "tokens/total": 9985696, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 151289 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.0109481830149889, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00029803148936480284, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0003, + "step": 331, + "tokens/total": 10016112, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 151757 + }, + { + "epoch": 1.296875, + "grad_norm": 0.011753400787711143, + "learning_rate": 3.734438472750619e-05, + "loss": 0.00021420676785055548, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00021, + "step": 332, + "tokens/total": 10046544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 152207 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.3032369613647461, + "learning_rate": 3.707773935267552e-05, + "loss": 0.004725611303001642, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00474, + "step": 333, + "tokens/total": 10077120, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 152646 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.024216441437602043, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0005056065274402499, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00051, + "step": 334, + "tokens/total": 10107600, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 153063 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.10152052342891693, + "learning_rate": 3.654669712344384e-05, + "loss": 0.001724423374980688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00173, + "step": 335, + "tokens/total": 10137712, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 153493 + }, + { + "epoch": 1.3125, + "grad_norm": 0.06117634102702141, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0013021706836298108, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0013, + "step": 336, + "tokens/total": 10168096, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 153945 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.022510197013616562, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.0003054860280826688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00031, + "step": 337, + "tokens/total": 10198720, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 154432 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.19209441542625427, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0020988616161048412, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0021, + "step": 338, + "tokens/total": 10228976, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 154881 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008805932477116585, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00020574698282871395, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00021, + "step": 339, + "tokens/total": 10259440, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 155306 + }, + { + "epoch": 1.328125, + "grad_norm": 0.01874561607837677, + "learning_rate": 3.5232722063479914e-05, + "loss": 0.0003597445320338011, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 340, + "tokens/total": 10289824, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 155719 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.005896273069083691, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010345762711949646, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10320144, + "tokens/train_per_sec_per_gpu": 36.3, + "tokens/trainable": 156209 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.010024541057646275, + "learning_rate": 3.471281389826491e-05, + "loss": 0.0001570192980580032, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00016, + "step": 342, + "tokens/total": 10350528, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 156656 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.10390307009220123, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0013503385707736015, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00135, + "step": 343, + "tokens/total": 10380816, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 157118 + }, + { + "epoch": 1.34375, + "grad_norm": 0.049824248999357224, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0005016025970689952, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0005, + "step": 344, + "tokens/total": 10411312, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 157562 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.011971387080848217, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.00018647368415258825, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 345, + "tokens/total": 10441696, + "tokens/train_per_sec_per_gpu": 35.7, + "tokens/trainable": 158003 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.020012276247143745, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.0004137884243391454, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00041, + "step": 346, + "tokens/total": 10472032, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 158419 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.1378275603055954, + "learning_rate": 3.342800532426873e-05, + "loss": 0.001066669705323875, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00107, + "step": 347, + "tokens/total": 10502176, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 158877 + }, + { + "epoch": 1.359375, + "grad_norm": 0.012419048696756363, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002563716552685946, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00026, + "step": 348, + "tokens/total": 10532896, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 159299 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0319787822663784, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0005879281088709831, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00059, + "step": 349, + "tokens/total": 10563024, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 159750 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.009861056692898273, + "learning_rate": 3.266780708475511e-05, + "loss": 0.00010148907313123345, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0001, + "step": 350, + "tokens/total": 10593536, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 160217 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.005884335841983557, + "learning_rate": 3.241625231705354e-05, + "loss": 0.00010095423203893006, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 351, + "tokens/total": 10623776, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 160675 + }, + { + "epoch": 1.375, + "grad_norm": 0.24247600138187408, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0036722179502248764, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00368, + "step": 352, + "tokens/total": 10654032, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 161132 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.231514112755758e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-352/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..6f7de269bfc864e94389bb4716c1bcae7bad8560 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6e12e316c4b40808347cd4df07ca3a1fc4190596cafc5155d40ea2a381b9cbda +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..4af13ede252b9eb4dd191999a67e41e1e711e7fb --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:87b2da364ff6d27ddfaa2bc1be957c48a94448d01c409a4a53226d6be598ede6 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..e7d74cf19667f47c685681f98aa1e9814222cca4 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f0f5fee0fb94cee084227608506532dac3dd7e8758af76be7848e1b7809a65f2 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c676299e10f0344839a2b08f1cc1e004410cac0 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8908527ed67c1624f630ff35c9d5ed61abd3040f0eae468424d1042bb8809a79 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..91f76032a8c84371c89ee6c6530dee08eb82c10f --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/tokens_state.json @@ -0,0 +1 @@ +{"total": 11625056, "trainable": 175750} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..645954cdf0185c5425148446c7287200c9201a1e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/trainer_state.json @@ -0,0 +1,5410 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.5, + "eval_steps": 500, + "global_step": 384, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.062361083924770355, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.0006638169870711863, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 321, + "tokens/total": 9712832, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 147139 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.009022507816553116, + "learning_rate": 4.004940255778431e-05, + "loss": 0.00017464160919189453, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00017, + "step": 322, + "tokens/total": 9743344, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 147589 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.38307738304138184, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0034184439573436975, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00342, + "step": 323, + "tokens/total": 9773760, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 148045 + }, + { + "epoch": 1.265625, + "grad_norm": 0.3203769028186798, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0051149846985936165, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00513, + "step": 324, + "tokens/total": 9804080, + "tokens/train_per_sec_per_gpu": 37.47, + "tokens/trainable": 148494 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.7488558292388916, + "learning_rate": 3.923084939213296e-05, + "loss": 0.01595677062869072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01608, + "step": 325, + "tokens/total": 9834400, + "tokens/train_per_sec_per_gpu": 36.84, + "tokens/trainable": 148957 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.009091824293136597, + "learning_rate": 3.895929566172861e-05, + "loss": 0.00014956737868487835, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 326, + "tokens/total": 9864704, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 149451 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.005110942758619785, + "learning_rate": 3.868840945050728e-05, + "loss": 0.00013913170550949872, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00014, + "step": 327, + "tokens/total": 9894656, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 149904 + }, + { + "epoch": 1.28125, + "grad_norm": 0.007466079201549292, + "learning_rate": 3.841820203114995e-05, + "loss": 0.00013769841461908072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00014, + "step": 328, + "tokens/total": 9924976, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 150378 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.16560006141662598, + "learning_rate": 3.814868464809027e-05, + "loss": 0.009101686999201775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00914, + "step": 329, + "tokens/total": 9955376, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 150853 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.01060777809470892, + "learning_rate": 3.787986851704667e-05, + "loss": 0.00019600678933784366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0002, + "step": 330, + "tokens/total": 9985696, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 151289 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.0109481830149889, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00029803148936480284, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0003, + "step": 331, + "tokens/total": 10016112, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 151757 + }, + { + "epoch": 1.296875, + "grad_norm": 0.011753400787711143, + "learning_rate": 3.734438472750619e-05, + "loss": 0.00021420676785055548, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00021, + "step": 332, + "tokens/total": 10046544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 152207 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.3032369613647461, + "learning_rate": 3.707773935267552e-05, + "loss": 0.004725611303001642, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00474, + "step": 333, + "tokens/total": 10077120, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 152646 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.024216441437602043, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0005056065274402499, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00051, + "step": 334, + "tokens/total": 10107600, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 153063 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.10152052342891693, + "learning_rate": 3.654669712344384e-05, + "loss": 0.001724423374980688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00173, + "step": 335, + "tokens/total": 10137712, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 153493 + }, + { + "epoch": 1.3125, + "grad_norm": 0.06117634102702141, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0013021706836298108, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0013, + "step": 336, + "tokens/total": 10168096, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 153945 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.022510197013616562, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.0003054860280826688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00031, + "step": 337, + "tokens/total": 10198720, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 154432 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.19209441542625427, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0020988616161048412, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0021, + "step": 338, + "tokens/total": 10228976, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 154881 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008805932477116585, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00020574698282871395, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00021, + "step": 339, + "tokens/total": 10259440, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 155306 + }, + { + "epoch": 1.328125, + "grad_norm": 0.01874561607837677, + "learning_rate": 3.5232722063479914e-05, + "loss": 0.0003597445320338011, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 340, + "tokens/total": 10289824, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 155719 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.005896273069083691, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010345762711949646, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10320144, + "tokens/train_per_sec_per_gpu": 36.3, + "tokens/trainable": 156209 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.010024541057646275, + "learning_rate": 3.471281389826491e-05, + "loss": 0.0001570192980580032, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00016, + "step": 342, + "tokens/total": 10350528, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 156656 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.10390307009220123, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0013503385707736015, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00135, + "step": 343, + "tokens/total": 10380816, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 157118 + }, + { + "epoch": 1.34375, + "grad_norm": 0.049824248999357224, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0005016025970689952, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0005, + "step": 344, + "tokens/total": 10411312, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 157562 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.011971387080848217, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.00018647368415258825, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 345, + "tokens/total": 10441696, + "tokens/train_per_sec_per_gpu": 35.7, + "tokens/trainable": 158003 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.020012276247143745, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.0004137884243391454, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00041, + "step": 346, + "tokens/total": 10472032, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 158419 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.1378275603055954, + "learning_rate": 3.342800532426873e-05, + "loss": 0.001066669705323875, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00107, + "step": 347, + "tokens/total": 10502176, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 158877 + }, + { + "epoch": 1.359375, + "grad_norm": 0.012419048696756363, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002563716552685946, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00026, + "step": 348, + "tokens/total": 10532896, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 159299 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0319787822663784, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0005879281088709831, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00059, + "step": 349, + "tokens/total": 10563024, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 159750 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.009861056692898273, + "learning_rate": 3.266780708475511e-05, + "loss": 0.00010148907313123345, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0001, + "step": 350, + "tokens/total": 10593536, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 160217 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.005884335841983557, + "learning_rate": 3.241625231705354e-05, + "loss": 0.00010095423203893006, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 351, + "tokens/total": 10623776, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 160675 + }, + { + "epoch": 1.375, + "grad_norm": 0.24247600138187408, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0036722179502248764, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00368, + "step": 352, + "tokens/total": 10654032, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 161132 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0033956863917410374, + "learning_rate": 3.191597261653475e-05, + "loss": 6.495713023468852e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00006, + "step": 353, + "tokens/total": 10684352, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 161585 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.8877756595611572, + "learning_rate": 3.166726850239794e-05, + "loss": 0.008384269662201405, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00842, + "step": 354, + "tokens/total": 10714816, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 162050 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00449851481243968, + "learning_rate": 3.141953535845912e-05, + "loss": 8.070325566222891e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 355, + "tokens/total": 10745296, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 162478 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0048718261532485485, + "learning_rate": 3.11727834939056e-05, + "loss": 7.578312943223864e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00008, + "step": 356, + "tokens/total": 10775296, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 162893 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.21484895050525665, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0030353569891303778, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00304, + "step": 357, + "tokens/total": 10805600, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 163347 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.07261230796575546, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.0004826942749787122, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00048, + "step": 358, + "tokens/total": 10835760, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 163796 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.022041240707039833, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00020454936020541936, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 359, + "tokens/total": 10866208, + "tokens/train_per_sec_per_gpu": 32.47, + "tokens/trainable": 164237 + }, + { + "epoch": 1.40625, + "grad_norm": 0.005708560813218355, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00012487702770158648, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00012, + "step": 360, + "tokens/total": 10896672, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 164684 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.02250049076974392, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.00025320789427496493, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00025, + "step": 361, + "tokens/total": 10926944, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 165146 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.013911883346736431, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00022382009774446487, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 362, + "tokens/total": 10957456, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 165632 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.2468303143978119, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.01283702440559864, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01292, + "step": 363, + "tokens/total": 10987696, + "tokens/train_per_sec_per_gpu": 34.66, + "tokens/trainable": 166080 + }, + { + "epoch": 1.421875, + "grad_norm": 0.008053003810346127, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00013117909838911146, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00013, + "step": 364, + "tokens/total": 11017904, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 166531 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.04743769019842148, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00024472299264743924, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00024, + "step": 365, + "tokens/total": 11048224, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 167019 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.008861004374921322, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00010162356193177402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 366, + "tokens/total": 11078544, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 167469 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.06953191757202148, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0006842018919996917, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00068, + "step": 367, + "tokens/total": 11109088, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 167914 + }, + { + "epoch": 1.4375, + "grad_norm": 0.10965435951948166, + "learning_rate": 2.829199644117484e-05, + "loss": 0.0006139783654361963, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00061, + "step": 368, + "tokens/total": 11139408, + "tokens/train_per_sec_per_gpu": 36.16, + "tokens/trainable": 168357 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.013294316828250885, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001493502495577559, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00015, + "step": 369, + "tokens/total": 11169920, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 168832 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.7463682889938354, + "learning_rate": 2.782696506053033e-05, + "loss": 0.013401923701167107, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01349, + "step": 370, + "tokens/total": 11200288, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 169333 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001093173515982926, + "learning_rate": 2.7596140715257824e-05, + "loss": 2.9635479222633876e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 371, + "tokens/total": 11230832, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 169774 + }, + { + "epoch": 1.453125, + "grad_norm": 0.1324324756860733, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0015135211870074272, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00151, + "step": 372, + "tokens/total": 11261264, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 170241 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.006301951594650745, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0001302927266806364, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00013, + "step": 373, + "tokens/total": 11291456, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 170706 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.011703136377036572, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00020681106252595782, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00021, + "step": 374, + "tokens/total": 11321952, + "tokens/train_per_sec_per_gpu": 32.53, + "tokens/trainable": 171182 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.015446359291672707, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0003060699673369527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00031, + "step": 375, + "tokens/total": 11352096, + "tokens/train_per_sec_per_gpu": 34.75, + "tokens/trainable": 171646 + }, + { + "epoch": 1.46875, + "grad_norm": 0.008862881921231747, + "learning_rate": 2.645931522709877e-05, + "loss": 0.00019395005074329674, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 376, + "tokens/total": 11382528, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 172063 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.03388524800539017, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0004883318324573338, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00049, + "step": 377, + "tokens/total": 11412832, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 172539 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.02546733431518078, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005159692373126745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11442864, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 172997 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.045234713703393936, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0004851966805290431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 379, + "tokens/total": 11473328, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 173481 + }, + { + "epoch": 1.484375, + "grad_norm": 0.011559339240193367, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00026303555932827294, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00026, + "step": 380, + "tokens/total": 11503504, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 173946 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.01947280764579773, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0004145601997151971, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00041, + "step": 381, + "tokens/total": 11533920, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 174417 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.03896784782409668, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.0005209866212680936, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00052, + "step": 382, + "tokens/total": 11564128, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 174872 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.012542990036308765, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00026536238146945834, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00027, + "step": 383, + "tokens/total": 11594672, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 175313 + }, + { + "epoch": 1.5, + "grad_norm": 0.23138198256492615, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.011072758585214615, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01113, + "step": 384, + "tokens/total": 11625056, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 175750 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 7.890604845712497e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-384/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..a9d1a8a8aa8811cf24f3d95b597eac83f0ac3f34 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:17b9b01936d3c08a85b19bc450a46a3c542df7b4ccdfbce4eb246466623e95b6 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..30084d491629fb9aa15a233325b658a1aad9eb10 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fd0d036472778760752be6def787b8de9dc1d5e2ec26c9c978ea57952b6712dd +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..345b032df8b32193c83974ed626ac7f41de68c96 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:64b177cec70f807704a17979be33619030d753ef926c4ec9201771f63712df24 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..20a981d8c8f57b8d2e18429ceb299414229ad6ac --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ed29f28c9b7651cfc581e35c7a694a1573d310ed4e8afccc5b01f2e33bedecc +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..07f9633e087f63746e9006e7c449c4d95d7c5db6 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/tokens_state.json @@ -0,0 +1 @@ +{"total": 12593328, "trainable": 190245} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..f68ce34b2836f7e234f1a8c832e011f42a2ef488 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/trainer_state.json @@ -0,0 +1,5858 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.625, + "eval_steps": 500, + "global_step": 416, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.062361083924770355, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.0006638169870711863, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 321, + "tokens/total": 9712832, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 147139 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.009022507816553116, + "learning_rate": 4.004940255778431e-05, + "loss": 0.00017464160919189453, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00017, + "step": 322, + "tokens/total": 9743344, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 147589 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.38307738304138184, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0034184439573436975, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00342, + "step": 323, + "tokens/total": 9773760, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 148045 + }, + { + "epoch": 1.265625, + "grad_norm": 0.3203769028186798, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0051149846985936165, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00513, + "step": 324, + "tokens/total": 9804080, + "tokens/train_per_sec_per_gpu": 37.47, + "tokens/trainable": 148494 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.7488558292388916, + "learning_rate": 3.923084939213296e-05, + "loss": 0.01595677062869072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01608, + "step": 325, + "tokens/total": 9834400, + "tokens/train_per_sec_per_gpu": 36.84, + "tokens/trainable": 148957 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.009091824293136597, + "learning_rate": 3.895929566172861e-05, + "loss": 0.00014956737868487835, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 326, + "tokens/total": 9864704, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 149451 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.005110942758619785, + "learning_rate": 3.868840945050728e-05, + "loss": 0.00013913170550949872, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00014, + "step": 327, + "tokens/total": 9894656, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 149904 + }, + { + "epoch": 1.28125, + "grad_norm": 0.007466079201549292, + "learning_rate": 3.841820203114995e-05, + "loss": 0.00013769841461908072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00014, + "step": 328, + "tokens/total": 9924976, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 150378 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.16560006141662598, + "learning_rate": 3.814868464809027e-05, + "loss": 0.009101686999201775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00914, + "step": 329, + "tokens/total": 9955376, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 150853 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.01060777809470892, + "learning_rate": 3.787986851704667e-05, + "loss": 0.00019600678933784366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0002, + "step": 330, + "tokens/total": 9985696, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 151289 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.0109481830149889, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00029803148936480284, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0003, + "step": 331, + "tokens/total": 10016112, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 151757 + }, + { + "epoch": 1.296875, + "grad_norm": 0.011753400787711143, + "learning_rate": 3.734438472750619e-05, + "loss": 0.00021420676785055548, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00021, + "step": 332, + "tokens/total": 10046544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 152207 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.3032369613647461, + "learning_rate": 3.707773935267552e-05, + "loss": 0.004725611303001642, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00474, + "step": 333, + "tokens/total": 10077120, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 152646 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.024216441437602043, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0005056065274402499, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00051, + "step": 334, + "tokens/total": 10107600, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 153063 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.10152052342891693, + "learning_rate": 3.654669712344384e-05, + "loss": 0.001724423374980688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00173, + "step": 335, + "tokens/total": 10137712, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 153493 + }, + { + "epoch": 1.3125, + "grad_norm": 0.06117634102702141, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0013021706836298108, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0013, + "step": 336, + "tokens/total": 10168096, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 153945 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.022510197013616562, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.0003054860280826688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00031, + "step": 337, + "tokens/total": 10198720, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 154432 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.19209441542625427, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0020988616161048412, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0021, + "step": 338, + "tokens/total": 10228976, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 154881 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008805932477116585, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00020574698282871395, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00021, + "step": 339, + "tokens/total": 10259440, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 155306 + }, + { + "epoch": 1.328125, + "grad_norm": 0.01874561607837677, + "learning_rate": 3.5232722063479914e-05, + "loss": 0.0003597445320338011, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 340, + "tokens/total": 10289824, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 155719 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.005896273069083691, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010345762711949646, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10320144, + "tokens/train_per_sec_per_gpu": 36.3, + "tokens/trainable": 156209 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.010024541057646275, + "learning_rate": 3.471281389826491e-05, + "loss": 0.0001570192980580032, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00016, + "step": 342, + "tokens/total": 10350528, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 156656 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.10390307009220123, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0013503385707736015, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00135, + "step": 343, + "tokens/total": 10380816, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 157118 + }, + { + "epoch": 1.34375, + "grad_norm": 0.049824248999357224, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0005016025970689952, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0005, + "step": 344, + "tokens/total": 10411312, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 157562 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.011971387080848217, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.00018647368415258825, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 345, + "tokens/total": 10441696, + "tokens/train_per_sec_per_gpu": 35.7, + "tokens/trainable": 158003 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.020012276247143745, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.0004137884243391454, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00041, + "step": 346, + "tokens/total": 10472032, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 158419 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.1378275603055954, + "learning_rate": 3.342800532426873e-05, + "loss": 0.001066669705323875, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00107, + "step": 347, + "tokens/total": 10502176, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 158877 + }, + { + "epoch": 1.359375, + "grad_norm": 0.012419048696756363, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002563716552685946, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00026, + "step": 348, + "tokens/total": 10532896, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 159299 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0319787822663784, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0005879281088709831, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00059, + "step": 349, + "tokens/total": 10563024, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 159750 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.009861056692898273, + "learning_rate": 3.266780708475511e-05, + "loss": 0.00010148907313123345, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0001, + "step": 350, + "tokens/total": 10593536, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 160217 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.005884335841983557, + "learning_rate": 3.241625231705354e-05, + "loss": 0.00010095423203893006, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 351, + "tokens/total": 10623776, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 160675 + }, + { + "epoch": 1.375, + "grad_norm": 0.24247600138187408, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0036722179502248764, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00368, + "step": 352, + "tokens/total": 10654032, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 161132 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0033956863917410374, + "learning_rate": 3.191597261653475e-05, + "loss": 6.495713023468852e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00006, + "step": 353, + "tokens/total": 10684352, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 161585 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.8877756595611572, + "learning_rate": 3.166726850239794e-05, + "loss": 0.008384269662201405, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00842, + "step": 354, + "tokens/total": 10714816, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 162050 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00449851481243968, + "learning_rate": 3.141953535845912e-05, + "loss": 8.070325566222891e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 355, + "tokens/total": 10745296, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 162478 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0048718261532485485, + "learning_rate": 3.11727834939056e-05, + "loss": 7.578312943223864e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00008, + "step": 356, + "tokens/total": 10775296, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 162893 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.21484895050525665, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0030353569891303778, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00304, + "step": 357, + "tokens/total": 10805600, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 163347 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.07261230796575546, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.0004826942749787122, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00048, + "step": 358, + "tokens/total": 10835760, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 163796 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.022041240707039833, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00020454936020541936, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 359, + "tokens/total": 10866208, + "tokens/train_per_sec_per_gpu": 32.47, + "tokens/trainable": 164237 + }, + { + "epoch": 1.40625, + "grad_norm": 0.005708560813218355, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00012487702770158648, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00012, + "step": 360, + "tokens/total": 10896672, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 164684 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.02250049076974392, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.00025320789427496493, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00025, + "step": 361, + "tokens/total": 10926944, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 165146 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.013911883346736431, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00022382009774446487, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 362, + "tokens/total": 10957456, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 165632 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.2468303143978119, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.01283702440559864, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01292, + "step": 363, + "tokens/total": 10987696, + "tokens/train_per_sec_per_gpu": 34.66, + "tokens/trainable": 166080 + }, + { + "epoch": 1.421875, + "grad_norm": 0.008053003810346127, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00013117909838911146, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00013, + "step": 364, + "tokens/total": 11017904, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 166531 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.04743769019842148, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00024472299264743924, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00024, + "step": 365, + "tokens/total": 11048224, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 167019 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.008861004374921322, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00010162356193177402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 366, + "tokens/total": 11078544, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 167469 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.06953191757202148, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0006842018919996917, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00068, + "step": 367, + "tokens/total": 11109088, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 167914 + }, + { + "epoch": 1.4375, + "grad_norm": 0.10965435951948166, + "learning_rate": 2.829199644117484e-05, + "loss": 0.0006139783654361963, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00061, + "step": 368, + "tokens/total": 11139408, + "tokens/train_per_sec_per_gpu": 36.16, + "tokens/trainable": 168357 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.013294316828250885, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001493502495577559, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00015, + "step": 369, + "tokens/total": 11169920, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 168832 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.7463682889938354, + "learning_rate": 2.782696506053033e-05, + "loss": 0.013401923701167107, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01349, + "step": 370, + "tokens/total": 11200288, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 169333 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001093173515982926, + "learning_rate": 2.7596140715257824e-05, + "loss": 2.9635479222633876e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 371, + "tokens/total": 11230832, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 169774 + }, + { + "epoch": 1.453125, + "grad_norm": 0.1324324756860733, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0015135211870074272, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00151, + "step": 372, + "tokens/total": 11261264, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 170241 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.006301951594650745, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0001302927266806364, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00013, + "step": 373, + "tokens/total": 11291456, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 170706 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.011703136377036572, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00020681106252595782, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00021, + "step": 374, + "tokens/total": 11321952, + "tokens/train_per_sec_per_gpu": 32.53, + "tokens/trainable": 171182 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.015446359291672707, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0003060699673369527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00031, + "step": 375, + "tokens/total": 11352096, + "tokens/train_per_sec_per_gpu": 34.75, + "tokens/trainable": 171646 + }, + { + "epoch": 1.46875, + "grad_norm": 0.008862881921231747, + "learning_rate": 2.645931522709877e-05, + "loss": 0.00019395005074329674, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 376, + "tokens/total": 11382528, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 172063 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.03388524800539017, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0004883318324573338, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00049, + "step": 377, + "tokens/total": 11412832, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 172539 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.02546733431518078, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005159692373126745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11442864, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 172997 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.045234713703393936, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0004851966805290431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 379, + "tokens/total": 11473328, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 173481 + }, + { + "epoch": 1.484375, + "grad_norm": 0.011559339240193367, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00026303555932827294, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00026, + "step": 380, + "tokens/total": 11503504, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 173946 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.01947280764579773, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0004145601997151971, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00041, + "step": 381, + "tokens/total": 11533920, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 174417 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.03896784782409668, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.0005209866212680936, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00052, + "step": 382, + "tokens/total": 11564128, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 174872 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.012542990036308765, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00026536238146945834, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00027, + "step": 383, + "tokens/total": 11594672, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 175313 + }, + { + "epoch": 1.5, + "grad_norm": 0.23138198256492615, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.011072758585214615, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01113, + "step": 384, + "tokens/total": 11625056, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 175750 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.5275915861129761, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.016837235540151596, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01698, + "step": 385, + "tokens/total": 11655488, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 176232 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.008643269538879395, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00011418825306463987, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 386, + "tokens/total": 11685968, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 176680 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.00684578949585557, + "learning_rate": 2.406442693028651e-05, + "loss": 0.0001287878112634644, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 387, + "tokens/total": 11716176, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 177144 + }, + { + "epoch": 1.515625, + "grad_norm": 0.07777013629674911, + "learning_rate": 2.3854255584458547e-05, + "loss": 0.001054117688909173, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00105, + "step": 388, + "tokens/total": 11746704, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 177630 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.021813517436385155, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.00045925029553472996, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00046, + "step": 389, + "tokens/total": 11777008, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 178079 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2766229808330536, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0063839266076684, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0064, + "step": 390, + "tokens/total": 11807168, + "tokens/train_per_sec_per_gpu": 36.99, + "tokens/trainable": 178552 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.5076630711555481, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.003972330130636692, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00398, + "step": 391, + "tokens/total": 11837776, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 179021 + }, + { + "epoch": 1.53125, + "grad_norm": 0.046427834779024124, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00048776037874631584, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 392, + "tokens/total": 11867968, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 179478 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.12148728966712952, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00316976523026824, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00317, + "step": 393, + "tokens/total": 11898320, + "tokens/train_per_sec_per_gpu": 33.27, + "tokens/trainable": 179910 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02432228811085224, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00029762828489765525, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0003, + "step": 394, + "tokens/total": 11928624, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 180361 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.01567767933011055, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.00036671021371148527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00037, + "step": 395, + "tokens/total": 11959088, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 180774 + }, + { + "epoch": 1.546875, + "grad_norm": 0.02068573608994484, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00030358770163729787, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0003, + "step": 396, + "tokens/total": 11989504, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 181246 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.2614375948905945, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00875169225037098, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00879, + "step": 397, + "tokens/total": 12019904, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 181734 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.037434663623571396, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00044167632586322725, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00044, + "step": 398, + "tokens/total": 12050384, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 182169 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.16984692215919495, + "learning_rate": 2.1629798596457056e-05, + "loss": 0.0007633643108420074, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00076, + "step": 399, + "tokens/total": 12080576, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 182641 + }, + { + "epoch": 1.5625, + "grad_norm": 0.17288610339164734, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.0036359927617013454, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00364, + "step": 400, + "tokens/total": 12110896, + "tokens/train_per_sec_per_gpu": 26.93, + "tokens/trainable": 183056 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.056392524391412735, + "learning_rate": 2.124308220868431e-05, + "loss": 0.0008229271625168622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00082, + "step": 401, + "tokens/total": 12140784, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 183504 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0365372970700264, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0007215660298243165, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00072, + "step": 402, + "tokens/total": 12171024, + "tokens/train_per_sec_per_gpu": 39.29, + "tokens/trainable": 184004 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.016568642109632492, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.00029834112501703203, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0003, + "step": 403, + "tokens/total": 12201184, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 184476 + }, + { + "epoch": 1.578125, + "grad_norm": 0.06592518836259842, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0007513194577768445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 404, + "tokens/total": 12231232, + "tokens/train_per_sec_per_gpu": 32.35, + "tokens/trainable": 184905 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.013345538638532162, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001536268973723054, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00015, + "step": 405, + "tokens/total": 12261280, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 185338 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.11538074165582657, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.001165087684057653, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00117, + "step": 406, + "tokens/total": 12291520, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 185792 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.01569611392915249, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00031081197084859014, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00031, + "step": 407, + "tokens/total": 12321824, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 186251 + }, + { + "epoch": 1.59375, + "grad_norm": 0.02253701724112034, + "learning_rate": 1.993423839463052e-05, + "loss": 0.00034153330489061773, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00034, + "step": 408, + "tokens/total": 12352176, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 186689 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.40578708052635193, + "learning_rate": 1.975303621298445e-05, + "loss": 0.005957326851785183, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00598, + "step": 409, + "tokens/total": 12382608, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 187121 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.15000925958156586, + "learning_rate": 1.957330080137385e-05, + "loss": 0.0013339307624846697, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00133, + "step": 410, + "tokens/total": 12413152, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 187568 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.006941859144717455, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00013002814375795424, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 411, + "tokens/total": 12443600, + "tokens/train_per_sec_per_gpu": 34.76, + "tokens/trainable": 188010 + }, + { + "epoch": 1.609375, + "grad_norm": 0.12384258210659027, + "learning_rate": 1.9218260145006073e-05, + "loss": 0.0010012347484007478, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.001, + "step": 412, + "tokens/total": 12471952, + "tokens/train_per_sec_per_gpu": 39.3, + "tokens/trainable": 188446 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.3418160378932953, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00740136718377471, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00743, + "step": 413, + "tokens/total": 12502480, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 188897 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.006758023519068956, + "learning_rate": 1.8869175523676064e-05, + "loss": 6.429506174754351e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00006, + "step": 414, + "tokens/total": 12532720, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 189366 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.19964486360549927, + "learning_rate": 1.869688492349885e-05, + "loss": 0.005143567454069853, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00516, + "step": 415, + "tokens/total": 12562848, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 189779 + }, + { + "epoch": 1.625, + "grad_norm": 0.16783277690410614, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007222781423479319, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 416, + "tokens/total": 12593328, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 190245 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 8.5478276354494e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-416/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..e5b0a62650c6f2144c7986274d5fb7cb7d53444b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f0b323f3fea45e02c2fabd61c429c6bdc140e7b3425cdbb0bc409168a0309a1f +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..07c303ca227e4020671e4cef80b605e39bc01e6d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5d60414e59da811244f04cd863580479c781287bf8ed3e16305b69047c61fb7f +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..21b7ac0c860280e1146ba7c80d4d3d9a8f44a239 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3aa9afe585253001856afc47614f0a2780fb092e3570494e6b46234b23895a17 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..2045249a1135f9f2d87c87c3e02e6227697d0cfa --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:457821a0c6da6ac211fae3b76339963abddd2e9839d13c28173a0924fca59503 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..fbb8f1f65f00d7862a2adf4aa9015475accabf30 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/tokens_state.json @@ -0,0 +1 @@ +{"total": 13563648, "trainable": 204886} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..2ef4d215479eea3850cfcef62a990aacb7a51452 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/trainer_state.json @@ -0,0 +1,6306 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.75, + "eval_steps": 500, + "global_step": 448, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.062361083924770355, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.0006638169870711863, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 321, + "tokens/total": 9712832, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 147139 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.009022507816553116, + "learning_rate": 4.004940255778431e-05, + "loss": 0.00017464160919189453, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00017, + "step": 322, + "tokens/total": 9743344, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 147589 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.38307738304138184, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0034184439573436975, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00342, + "step": 323, + "tokens/total": 9773760, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 148045 + }, + { + "epoch": 1.265625, + "grad_norm": 0.3203769028186798, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0051149846985936165, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00513, + "step": 324, + "tokens/total": 9804080, + "tokens/train_per_sec_per_gpu": 37.47, + "tokens/trainable": 148494 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.7488558292388916, + "learning_rate": 3.923084939213296e-05, + "loss": 0.01595677062869072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01608, + "step": 325, + "tokens/total": 9834400, + "tokens/train_per_sec_per_gpu": 36.84, + "tokens/trainable": 148957 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.009091824293136597, + "learning_rate": 3.895929566172861e-05, + "loss": 0.00014956737868487835, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 326, + "tokens/total": 9864704, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 149451 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.005110942758619785, + "learning_rate": 3.868840945050728e-05, + "loss": 0.00013913170550949872, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00014, + "step": 327, + "tokens/total": 9894656, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 149904 + }, + { + "epoch": 1.28125, + "grad_norm": 0.007466079201549292, + "learning_rate": 3.841820203114995e-05, + "loss": 0.00013769841461908072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00014, + "step": 328, + "tokens/total": 9924976, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 150378 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.16560006141662598, + "learning_rate": 3.814868464809027e-05, + "loss": 0.009101686999201775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00914, + "step": 329, + "tokens/total": 9955376, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 150853 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.01060777809470892, + "learning_rate": 3.787986851704667e-05, + "loss": 0.00019600678933784366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0002, + "step": 330, + "tokens/total": 9985696, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 151289 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.0109481830149889, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00029803148936480284, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0003, + "step": 331, + "tokens/total": 10016112, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 151757 + }, + { + "epoch": 1.296875, + "grad_norm": 0.011753400787711143, + "learning_rate": 3.734438472750619e-05, + "loss": 0.00021420676785055548, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00021, + "step": 332, + "tokens/total": 10046544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 152207 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.3032369613647461, + "learning_rate": 3.707773935267552e-05, + "loss": 0.004725611303001642, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00474, + "step": 333, + "tokens/total": 10077120, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 152646 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.024216441437602043, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0005056065274402499, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00051, + "step": 334, + "tokens/total": 10107600, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 153063 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.10152052342891693, + "learning_rate": 3.654669712344384e-05, + "loss": 0.001724423374980688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00173, + "step": 335, + "tokens/total": 10137712, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 153493 + }, + { + "epoch": 1.3125, + "grad_norm": 0.06117634102702141, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0013021706836298108, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0013, + "step": 336, + "tokens/total": 10168096, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 153945 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.022510197013616562, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.0003054860280826688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00031, + "step": 337, + "tokens/total": 10198720, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 154432 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.19209441542625427, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0020988616161048412, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0021, + "step": 338, + "tokens/total": 10228976, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 154881 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008805932477116585, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00020574698282871395, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00021, + "step": 339, + "tokens/total": 10259440, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 155306 + }, + { + "epoch": 1.328125, + "grad_norm": 0.01874561607837677, + "learning_rate": 3.5232722063479914e-05, + "loss": 0.0003597445320338011, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 340, + "tokens/total": 10289824, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 155719 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.005896273069083691, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010345762711949646, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10320144, + "tokens/train_per_sec_per_gpu": 36.3, + "tokens/trainable": 156209 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.010024541057646275, + "learning_rate": 3.471281389826491e-05, + "loss": 0.0001570192980580032, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00016, + "step": 342, + "tokens/total": 10350528, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 156656 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.10390307009220123, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0013503385707736015, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00135, + "step": 343, + "tokens/total": 10380816, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 157118 + }, + { + "epoch": 1.34375, + "grad_norm": 0.049824248999357224, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0005016025970689952, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0005, + "step": 344, + "tokens/total": 10411312, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 157562 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.011971387080848217, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.00018647368415258825, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 345, + "tokens/total": 10441696, + "tokens/train_per_sec_per_gpu": 35.7, + "tokens/trainable": 158003 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.020012276247143745, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.0004137884243391454, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00041, + "step": 346, + "tokens/total": 10472032, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 158419 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.1378275603055954, + "learning_rate": 3.342800532426873e-05, + "loss": 0.001066669705323875, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00107, + "step": 347, + "tokens/total": 10502176, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 158877 + }, + { + "epoch": 1.359375, + "grad_norm": 0.012419048696756363, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002563716552685946, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00026, + "step": 348, + "tokens/total": 10532896, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 159299 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0319787822663784, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0005879281088709831, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00059, + "step": 349, + "tokens/total": 10563024, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 159750 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.009861056692898273, + "learning_rate": 3.266780708475511e-05, + "loss": 0.00010148907313123345, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0001, + "step": 350, + "tokens/total": 10593536, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 160217 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.005884335841983557, + "learning_rate": 3.241625231705354e-05, + "loss": 0.00010095423203893006, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 351, + "tokens/total": 10623776, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 160675 + }, + { + "epoch": 1.375, + "grad_norm": 0.24247600138187408, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0036722179502248764, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00368, + "step": 352, + "tokens/total": 10654032, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 161132 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0033956863917410374, + "learning_rate": 3.191597261653475e-05, + "loss": 6.495713023468852e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00006, + "step": 353, + "tokens/total": 10684352, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 161585 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.8877756595611572, + "learning_rate": 3.166726850239794e-05, + "loss": 0.008384269662201405, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00842, + "step": 354, + "tokens/total": 10714816, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 162050 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00449851481243968, + "learning_rate": 3.141953535845912e-05, + "loss": 8.070325566222891e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 355, + "tokens/total": 10745296, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 162478 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0048718261532485485, + "learning_rate": 3.11727834939056e-05, + "loss": 7.578312943223864e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00008, + "step": 356, + "tokens/total": 10775296, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 162893 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.21484895050525665, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0030353569891303778, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00304, + "step": 357, + "tokens/total": 10805600, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 163347 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.07261230796575546, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.0004826942749787122, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00048, + "step": 358, + "tokens/total": 10835760, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 163796 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.022041240707039833, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00020454936020541936, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 359, + "tokens/total": 10866208, + "tokens/train_per_sec_per_gpu": 32.47, + "tokens/trainable": 164237 + }, + { + "epoch": 1.40625, + "grad_norm": 0.005708560813218355, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00012487702770158648, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00012, + "step": 360, + "tokens/total": 10896672, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 164684 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.02250049076974392, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.00025320789427496493, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00025, + "step": 361, + "tokens/total": 10926944, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 165146 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.013911883346736431, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00022382009774446487, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 362, + "tokens/total": 10957456, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 165632 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.2468303143978119, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.01283702440559864, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01292, + "step": 363, + "tokens/total": 10987696, + "tokens/train_per_sec_per_gpu": 34.66, + "tokens/trainable": 166080 + }, + { + "epoch": 1.421875, + "grad_norm": 0.008053003810346127, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00013117909838911146, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00013, + "step": 364, + "tokens/total": 11017904, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 166531 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.04743769019842148, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00024472299264743924, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00024, + "step": 365, + "tokens/total": 11048224, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 167019 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.008861004374921322, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00010162356193177402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 366, + "tokens/total": 11078544, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 167469 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.06953191757202148, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0006842018919996917, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00068, + "step": 367, + "tokens/total": 11109088, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 167914 + }, + { + "epoch": 1.4375, + "grad_norm": 0.10965435951948166, + "learning_rate": 2.829199644117484e-05, + "loss": 0.0006139783654361963, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00061, + "step": 368, + "tokens/total": 11139408, + "tokens/train_per_sec_per_gpu": 36.16, + "tokens/trainable": 168357 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.013294316828250885, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001493502495577559, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00015, + "step": 369, + "tokens/total": 11169920, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 168832 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.7463682889938354, + "learning_rate": 2.782696506053033e-05, + "loss": 0.013401923701167107, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01349, + "step": 370, + "tokens/total": 11200288, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 169333 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001093173515982926, + "learning_rate": 2.7596140715257824e-05, + "loss": 2.9635479222633876e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 371, + "tokens/total": 11230832, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 169774 + }, + { + "epoch": 1.453125, + "grad_norm": 0.1324324756860733, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0015135211870074272, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00151, + "step": 372, + "tokens/total": 11261264, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 170241 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.006301951594650745, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0001302927266806364, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00013, + "step": 373, + "tokens/total": 11291456, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 170706 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.011703136377036572, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00020681106252595782, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00021, + "step": 374, + "tokens/total": 11321952, + "tokens/train_per_sec_per_gpu": 32.53, + "tokens/trainable": 171182 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.015446359291672707, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0003060699673369527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00031, + "step": 375, + "tokens/total": 11352096, + "tokens/train_per_sec_per_gpu": 34.75, + "tokens/trainable": 171646 + }, + { + "epoch": 1.46875, + "grad_norm": 0.008862881921231747, + "learning_rate": 2.645931522709877e-05, + "loss": 0.00019395005074329674, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 376, + "tokens/total": 11382528, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 172063 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.03388524800539017, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0004883318324573338, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00049, + "step": 377, + "tokens/total": 11412832, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 172539 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.02546733431518078, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005159692373126745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11442864, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 172997 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.045234713703393936, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0004851966805290431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 379, + "tokens/total": 11473328, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 173481 + }, + { + "epoch": 1.484375, + "grad_norm": 0.011559339240193367, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00026303555932827294, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00026, + "step": 380, + "tokens/total": 11503504, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 173946 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.01947280764579773, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0004145601997151971, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00041, + "step": 381, + "tokens/total": 11533920, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 174417 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.03896784782409668, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.0005209866212680936, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00052, + "step": 382, + "tokens/total": 11564128, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 174872 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.012542990036308765, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00026536238146945834, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00027, + "step": 383, + "tokens/total": 11594672, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 175313 + }, + { + "epoch": 1.5, + "grad_norm": 0.23138198256492615, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.011072758585214615, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01113, + "step": 384, + "tokens/total": 11625056, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 175750 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.5275915861129761, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.016837235540151596, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01698, + "step": 385, + "tokens/total": 11655488, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 176232 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.008643269538879395, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00011418825306463987, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 386, + "tokens/total": 11685968, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 176680 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.00684578949585557, + "learning_rate": 2.406442693028651e-05, + "loss": 0.0001287878112634644, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 387, + "tokens/total": 11716176, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 177144 + }, + { + "epoch": 1.515625, + "grad_norm": 0.07777013629674911, + "learning_rate": 2.3854255584458547e-05, + "loss": 0.001054117688909173, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00105, + "step": 388, + "tokens/total": 11746704, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 177630 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.021813517436385155, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.00045925029553472996, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00046, + "step": 389, + "tokens/total": 11777008, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 178079 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2766229808330536, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0063839266076684, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0064, + "step": 390, + "tokens/total": 11807168, + "tokens/train_per_sec_per_gpu": 36.99, + "tokens/trainable": 178552 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.5076630711555481, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.003972330130636692, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00398, + "step": 391, + "tokens/total": 11837776, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 179021 + }, + { + "epoch": 1.53125, + "grad_norm": 0.046427834779024124, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00048776037874631584, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 392, + "tokens/total": 11867968, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 179478 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.12148728966712952, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00316976523026824, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00317, + "step": 393, + "tokens/total": 11898320, + "tokens/train_per_sec_per_gpu": 33.27, + "tokens/trainable": 179910 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02432228811085224, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00029762828489765525, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0003, + "step": 394, + "tokens/total": 11928624, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 180361 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.01567767933011055, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.00036671021371148527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00037, + "step": 395, + "tokens/total": 11959088, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 180774 + }, + { + "epoch": 1.546875, + "grad_norm": 0.02068573608994484, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00030358770163729787, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0003, + "step": 396, + "tokens/total": 11989504, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 181246 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.2614375948905945, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00875169225037098, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00879, + "step": 397, + "tokens/total": 12019904, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 181734 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.037434663623571396, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00044167632586322725, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00044, + "step": 398, + "tokens/total": 12050384, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 182169 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.16984692215919495, + "learning_rate": 2.1629798596457056e-05, + "loss": 0.0007633643108420074, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00076, + "step": 399, + "tokens/total": 12080576, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 182641 + }, + { + "epoch": 1.5625, + "grad_norm": 0.17288610339164734, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.0036359927617013454, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00364, + "step": 400, + "tokens/total": 12110896, + "tokens/train_per_sec_per_gpu": 26.93, + "tokens/trainable": 183056 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.056392524391412735, + "learning_rate": 2.124308220868431e-05, + "loss": 0.0008229271625168622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00082, + "step": 401, + "tokens/total": 12140784, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 183504 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0365372970700264, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0007215660298243165, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00072, + "step": 402, + "tokens/total": 12171024, + "tokens/train_per_sec_per_gpu": 39.29, + "tokens/trainable": 184004 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.016568642109632492, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.00029834112501703203, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0003, + "step": 403, + "tokens/total": 12201184, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 184476 + }, + { + "epoch": 1.578125, + "grad_norm": 0.06592518836259842, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0007513194577768445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 404, + "tokens/total": 12231232, + "tokens/train_per_sec_per_gpu": 32.35, + "tokens/trainable": 184905 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.013345538638532162, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001536268973723054, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00015, + "step": 405, + "tokens/total": 12261280, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 185338 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.11538074165582657, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.001165087684057653, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00117, + "step": 406, + "tokens/total": 12291520, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 185792 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.01569611392915249, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00031081197084859014, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00031, + "step": 407, + "tokens/total": 12321824, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 186251 + }, + { + "epoch": 1.59375, + "grad_norm": 0.02253701724112034, + "learning_rate": 1.993423839463052e-05, + "loss": 0.00034153330489061773, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00034, + "step": 408, + "tokens/total": 12352176, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 186689 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.40578708052635193, + "learning_rate": 1.975303621298445e-05, + "loss": 0.005957326851785183, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00598, + "step": 409, + "tokens/total": 12382608, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 187121 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.15000925958156586, + "learning_rate": 1.957330080137385e-05, + "loss": 0.0013339307624846697, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00133, + "step": 410, + "tokens/total": 12413152, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 187568 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.006941859144717455, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00013002814375795424, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 411, + "tokens/total": 12443600, + "tokens/train_per_sec_per_gpu": 34.76, + "tokens/trainable": 188010 + }, + { + "epoch": 1.609375, + "grad_norm": 0.12384258210659027, + "learning_rate": 1.9218260145006073e-05, + "loss": 0.0010012347484007478, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.001, + "step": 412, + "tokens/total": 12471952, + "tokens/train_per_sec_per_gpu": 39.3, + "tokens/trainable": 188446 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.3418160378932953, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00740136718377471, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00743, + "step": 413, + "tokens/total": 12502480, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 188897 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.006758023519068956, + "learning_rate": 1.8869175523676064e-05, + "loss": 6.429506174754351e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00006, + "step": 414, + "tokens/total": 12532720, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 189366 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.19964486360549927, + "learning_rate": 1.869688492349885e-05, + "loss": 0.005143567454069853, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00516, + "step": 415, + "tokens/total": 12562848, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 189779 + }, + { + "epoch": 1.625, + "grad_norm": 0.16783277690410614, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007222781423479319, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 416, + "tokens/total": 12593328, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 190245 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.018791699782013893, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.0002913455246016383, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00029, + "step": 417, + "tokens/total": 12623760, + "tokens/train_per_sec_per_gpu": 35.59, + "tokens/trainable": 190737 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.09821721911430359, + "learning_rate": 1.8189105812005714e-05, + "loss": 0.0005217839498072863, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00052, + "step": 418, + "tokens/total": 12654272, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 191155 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.071963369846344, + "learning_rate": 1.802290048317732e-05, + "loss": 0.0007684896700084209, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00077, + "step": 419, + "tokens/total": 12684544, + "tokens/train_per_sec_per_gpu": 31.98, + "tokens/trainable": 191577 + }, + { + "epoch": 1.640625, + "grad_norm": 0.041365377604961395, + "learning_rate": 1.785823392239424e-05, + "loss": 0.0007164644775912166, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00072, + "step": 420, + "tokens/total": 12715008, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 192051 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.01246409397572279, + "learning_rate": 1.7695112982104225e-05, + "loss": 0.0002488879836164415, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00025, + "step": 421, + "tokens/total": 12745232, + "tokens/train_per_sec_per_gpu": 31.77, + "tokens/trainable": 192500 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.00797255989164114, + "learning_rate": 1.7533544450435433e-05, + "loss": 8.724055078346282e-05, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 422, + "tokens/total": 12775824, + "tokens/train_per_sec_per_gpu": 33.23, + "tokens/trainable": 192955 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.029961727559566498, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.00036373038892634213, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00036, + "step": 423, + "tokens/total": 12806240, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 193462 + }, + { + "epoch": 1.65625, + "grad_norm": 0.11438705027103424, + "learning_rate": 1.721509144218405e-05, + "loss": 0.0008654020493850112, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00087, + "step": 424, + "tokens/total": 12836512, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 193943 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.11104747653007507, + "learning_rate": 1.705822021773101e-05, + "loss": 0.0019330887589603662, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00193, + "step": 425, + "tokens/total": 12866704, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 194360 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.005441099405288696, + "learning_rate": 1.69029279056068e-05, + "loss": 0.00012030347716063261, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00012, + "step": 426, + "tokens/total": 12896912, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 194848 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.010912942700088024, + "learning_rate": 1.6749220968158415e-05, + "loss": 0.00012806981976609677, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00013, + "step": 427, + "tokens/total": 12927248, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 195314 + }, + { + "epoch": 1.671875, + "grad_norm": 0.007124146446585655, + "learning_rate": 1.659710580175893e-05, + "loss": 9.905424667522311e-05, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0001, + "step": 428, + "tokens/total": 12957632, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 195732 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.23617680370807648, + "learning_rate": 1.644658873654133e-05, + "loss": 0.0024689119309186935, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00247, + "step": 429, + "tokens/total": 12987824, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 196197 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.29176223278045654, + "learning_rate": 1.629767603613508e-05, + "loss": 0.005283118691295385, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0053, + "step": 430, + "tokens/total": 13018208, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 196678 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.030829036608338356, + "learning_rate": 1.615037389740547e-05, + "loss": 0.0004377455043140799, + "memory/device_reserved (GiB)": 36.36, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00044, + "step": 431, + "tokens/total": 13048800, + "tokens/train_per_sec_per_gpu": 31.52, + "tokens/trainable": 197109 + }, + { + "epoch": 1.6875, + "grad_norm": 0.023997988551855087, + "learning_rate": 1.600468845019576e-05, + "loss": 0.0002723708748817444, + "memory/device_reserved (GiB)": 36.36, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00027, + "step": 432, + "tokens/total": 13079392, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 197607 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.03983060643076897, + "learning_rate": 1.5860625757072092e-05, + "loss": 0.0006240076618269086, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00062, + "step": 433, + "tokens/total": 13109920, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 198066 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.03856171295046806, + "learning_rate": 1.571819181307116e-05, + "loss": 0.0005622098105959594, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00056, + "step": 434, + "tokens/total": 13140064, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 198484 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.07453715056180954, + "learning_rate": 1.557739254545075e-05, + "loss": 0.0008548495243303478, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00086, + "step": 435, + "tokens/total": 13170656, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 198957 + }, + { + "epoch": 1.703125, + "grad_norm": 0.0035155205987393856, + "learning_rate": 1.543823381344311e-05, + "loss": 5.2120005420874804e-05, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 436, + "tokens/total": 13201056, + "tokens/train_per_sec_per_gpu": 33.78, + "tokens/trainable": 199405 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.9568848013877869, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.00043475639540702105, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00043, + "step": 437, + "tokens/total": 13231584, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 199888 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.010019529610872269, + "learning_rate": 1.5164861051607254e-05, + "loss": 0.000111720735731069, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 438, + "tokens/total": 13261888, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 200396 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.008281780406832695, + "learning_rate": 1.5030658397935521e-05, + "loss": 0.00017071102047339082, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 439, + "tokens/total": 13291936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 200843 + }, + { + "epoch": 1.71875, + "grad_norm": 0.06077976152300835, + "learning_rate": 1.4898119031716104e-05, + "loss": 0.0007504730019718409, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 440, + "tokens/total": 13322224, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 201315 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.12958590686321259, + "learning_rate": 1.476724846845306e-05, + "loss": 0.001083776238374412, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00108, + "step": 441, + "tokens/total": 13352640, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 201782 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.04559873417019844, + "learning_rate": 1.463805215420471e-05, + "loss": 0.0005228023510426283, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00052, + "step": 442, + "tokens/total": 13382560, + "tokens/train_per_sec_per_gpu": 31.27, + "tokens/trainable": 202222 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.02645108290016651, + "learning_rate": 1.451053546535705e-05, + "loss": 0.0003615481255110353, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00036, + "step": 443, + "tokens/total": 13412976, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 202660 + }, + { + "epoch": 1.734375, + "grad_norm": 0.021894006058573723, + "learning_rate": 1.438470370840001e-05, + "loss": 0.000229351528105326, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00023, + "step": 444, + "tokens/total": 13442720, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 203068 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.1398804783821106, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.0013714064843952656, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00137, + "step": 445, + "tokens/total": 13472928, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 203527 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.024557381868362427, + "learning_rate": 1.413811586531508e-05, + "loss": 0.00023480730305891484, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00023, + "step": 446, + "tokens/total": 13502992, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 203994 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.021804476156830788, + "learning_rate": 1.4017370040713884e-05, + "loss": 0.00022578673087991774, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00023, + "step": 447, + "tokens/total": 13533184, + "tokens/train_per_sec_per_gpu": 38.06, + "tokens/trainable": 204430 + }, + { + "epoch": 1.75, + "grad_norm": 0.23265990614891052, + "learning_rate": 1.3898329670629645e-05, + "loss": 0.002107386477291584, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00211, + "step": 448, + "tokens/total": 13563648, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 204886 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.206440522466181e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-448/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..cedb4217f2daba67f74512b1af2f1d9daabdd9a6 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e767f38d93516d2a91d9db9148bc7a639c69596a4261e7cf09dc557e23edead2 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..00caa4437316d85d503bd0290763e3de7695b344 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:598167917e6ecb20f95ae116c75b2a2dbff7f53e6763d0c4a52cea2e1b44bc94 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..4087200b08ac7eba3c18e36898b0737bd1db485c --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7e112da89198d46f80dee18f0562406093076519aaf838b7006fe23784d97abc +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..ce1c4e63a6c6f8be8c137d3add4eda184d59a043 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b68903e42777cffa699a4ceb5e57a17a53fd50d8cdf046737093de36719dc01 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..4324a8093505295b536afae72a2e061ffeca7fd0 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/tokens_state.json @@ -0,0 +1 @@ +{"total": 14528976, "trainable": 219411} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7da1fe57a2057bf47590073db30af77f2909cf9d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/trainer_state.json @@ -0,0 +1,6754 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.875, + "eval_steps": 500, + "global_step": 480, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.062361083924770355, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.0006638169870711863, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 321, + "tokens/total": 9712832, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 147139 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.009022507816553116, + "learning_rate": 4.004940255778431e-05, + "loss": 0.00017464160919189453, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00017, + "step": 322, + "tokens/total": 9743344, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 147589 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.38307738304138184, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0034184439573436975, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00342, + "step": 323, + "tokens/total": 9773760, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 148045 + }, + { + "epoch": 1.265625, + "grad_norm": 0.3203769028186798, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0051149846985936165, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00513, + "step": 324, + "tokens/total": 9804080, + "tokens/train_per_sec_per_gpu": 37.47, + "tokens/trainable": 148494 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.7488558292388916, + "learning_rate": 3.923084939213296e-05, + "loss": 0.01595677062869072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01608, + "step": 325, + "tokens/total": 9834400, + "tokens/train_per_sec_per_gpu": 36.84, + "tokens/trainable": 148957 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.009091824293136597, + "learning_rate": 3.895929566172861e-05, + "loss": 0.00014956737868487835, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 326, + "tokens/total": 9864704, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 149451 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.005110942758619785, + "learning_rate": 3.868840945050728e-05, + "loss": 0.00013913170550949872, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00014, + "step": 327, + "tokens/total": 9894656, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 149904 + }, + { + "epoch": 1.28125, + "grad_norm": 0.007466079201549292, + "learning_rate": 3.841820203114995e-05, + "loss": 0.00013769841461908072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00014, + "step": 328, + "tokens/total": 9924976, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 150378 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.16560006141662598, + "learning_rate": 3.814868464809027e-05, + "loss": 0.009101686999201775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00914, + "step": 329, + "tokens/total": 9955376, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 150853 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.01060777809470892, + "learning_rate": 3.787986851704667e-05, + "loss": 0.00019600678933784366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0002, + "step": 330, + "tokens/total": 9985696, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 151289 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.0109481830149889, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00029803148936480284, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0003, + "step": 331, + "tokens/total": 10016112, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 151757 + }, + { + "epoch": 1.296875, + "grad_norm": 0.011753400787711143, + "learning_rate": 3.734438472750619e-05, + "loss": 0.00021420676785055548, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00021, + "step": 332, + "tokens/total": 10046544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 152207 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.3032369613647461, + "learning_rate": 3.707773935267552e-05, + "loss": 0.004725611303001642, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00474, + "step": 333, + "tokens/total": 10077120, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 152646 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.024216441437602043, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0005056065274402499, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00051, + "step": 334, + "tokens/total": 10107600, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 153063 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.10152052342891693, + "learning_rate": 3.654669712344384e-05, + "loss": 0.001724423374980688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00173, + "step": 335, + "tokens/total": 10137712, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 153493 + }, + { + "epoch": 1.3125, + "grad_norm": 0.06117634102702141, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0013021706836298108, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0013, + "step": 336, + "tokens/total": 10168096, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 153945 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.022510197013616562, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.0003054860280826688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00031, + "step": 337, + "tokens/total": 10198720, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 154432 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.19209441542625427, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0020988616161048412, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0021, + "step": 338, + "tokens/total": 10228976, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 154881 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008805932477116585, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00020574698282871395, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00021, + "step": 339, + "tokens/total": 10259440, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 155306 + }, + { + "epoch": 1.328125, + "grad_norm": 0.01874561607837677, + "learning_rate": 3.5232722063479914e-05, + "loss": 0.0003597445320338011, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 340, + "tokens/total": 10289824, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 155719 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.005896273069083691, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010345762711949646, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10320144, + "tokens/train_per_sec_per_gpu": 36.3, + "tokens/trainable": 156209 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.010024541057646275, + "learning_rate": 3.471281389826491e-05, + "loss": 0.0001570192980580032, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00016, + "step": 342, + "tokens/total": 10350528, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 156656 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.10390307009220123, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0013503385707736015, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00135, + "step": 343, + "tokens/total": 10380816, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 157118 + }, + { + "epoch": 1.34375, + "grad_norm": 0.049824248999357224, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0005016025970689952, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0005, + "step": 344, + "tokens/total": 10411312, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 157562 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.011971387080848217, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.00018647368415258825, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 345, + "tokens/total": 10441696, + "tokens/train_per_sec_per_gpu": 35.7, + "tokens/trainable": 158003 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.020012276247143745, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.0004137884243391454, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00041, + "step": 346, + "tokens/total": 10472032, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 158419 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.1378275603055954, + "learning_rate": 3.342800532426873e-05, + "loss": 0.001066669705323875, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00107, + "step": 347, + "tokens/total": 10502176, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 158877 + }, + { + "epoch": 1.359375, + "grad_norm": 0.012419048696756363, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002563716552685946, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00026, + "step": 348, + "tokens/total": 10532896, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 159299 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0319787822663784, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0005879281088709831, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00059, + "step": 349, + "tokens/total": 10563024, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 159750 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.009861056692898273, + "learning_rate": 3.266780708475511e-05, + "loss": 0.00010148907313123345, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0001, + "step": 350, + "tokens/total": 10593536, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 160217 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.005884335841983557, + "learning_rate": 3.241625231705354e-05, + "loss": 0.00010095423203893006, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 351, + "tokens/total": 10623776, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 160675 + }, + { + "epoch": 1.375, + "grad_norm": 0.24247600138187408, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0036722179502248764, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00368, + "step": 352, + "tokens/total": 10654032, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 161132 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0033956863917410374, + "learning_rate": 3.191597261653475e-05, + "loss": 6.495713023468852e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00006, + "step": 353, + "tokens/total": 10684352, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 161585 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.8877756595611572, + "learning_rate": 3.166726850239794e-05, + "loss": 0.008384269662201405, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00842, + "step": 354, + "tokens/total": 10714816, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 162050 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00449851481243968, + "learning_rate": 3.141953535845912e-05, + "loss": 8.070325566222891e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 355, + "tokens/total": 10745296, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 162478 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0048718261532485485, + "learning_rate": 3.11727834939056e-05, + "loss": 7.578312943223864e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00008, + "step": 356, + "tokens/total": 10775296, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 162893 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.21484895050525665, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0030353569891303778, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00304, + "step": 357, + "tokens/total": 10805600, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 163347 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.07261230796575546, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.0004826942749787122, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00048, + "step": 358, + "tokens/total": 10835760, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 163796 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.022041240707039833, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00020454936020541936, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 359, + "tokens/total": 10866208, + "tokens/train_per_sec_per_gpu": 32.47, + "tokens/trainable": 164237 + }, + { + "epoch": 1.40625, + "grad_norm": 0.005708560813218355, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00012487702770158648, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00012, + "step": 360, + "tokens/total": 10896672, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 164684 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.02250049076974392, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.00025320789427496493, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00025, + "step": 361, + "tokens/total": 10926944, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 165146 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.013911883346736431, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00022382009774446487, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 362, + "tokens/total": 10957456, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 165632 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.2468303143978119, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.01283702440559864, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01292, + "step": 363, + "tokens/total": 10987696, + "tokens/train_per_sec_per_gpu": 34.66, + "tokens/trainable": 166080 + }, + { + "epoch": 1.421875, + "grad_norm": 0.008053003810346127, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00013117909838911146, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00013, + "step": 364, + "tokens/total": 11017904, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 166531 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.04743769019842148, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00024472299264743924, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00024, + "step": 365, + "tokens/total": 11048224, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 167019 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.008861004374921322, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00010162356193177402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 366, + "tokens/total": 11078544, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 167469 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.06953191757202148, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0006842018919996917, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00068, + "step": 367, + "tokens/total": 11109088, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 167914 + }, + { + "epoch": 1.4375, + "grad_norm": 0.10965435951948166, + "learning_rate": 2.829199644117484e-05, + "loss": 0.0006139783654361963, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00061, + "step": 368, + "tokens/total": 11139408, + "tokens/train_per_sec_per_gpu": 36.16, + "tokens/trainable": 168357 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.013294316828250885, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001493502495577559, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00015, + "step": 369, + "tokens/total": 11169920, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 168832 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.7463682889938354, + "learning_rate": 2.782696506053033e-05, + "loss": 0.013401923701167107, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01349, + "step": 370, + "tokens/total": 11200288, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 169333 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001093173515982926, + "learning_rate": 2.7596140715257824e-05, + "loss": 2.9635479222633876e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 371, + "tokens/total": 11230832, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 169774 + }, + { + "epoch": 1.453125, + "grad_norm": 0.1324324756860733, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0015135211870074272, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00151, + "step": 372, + "tokens/total": 11261264, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 170241 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.006301951594650745, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0001302927266806364, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00013, + "step": 373, + "tokens/total": 11291456, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 170706 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.011703136377036572, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00020681106252595782, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00021, + "step": 374, + "tokens/total": 11321952, + "tokens/train_per_sec_per_gpu": 32.53, + "tokens/trainable": 171182 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.015446359291672707, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0003060699673369527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00031, + "step": 375, + "tokens/total": 11352096, + "tokens/train_per_sec_per_gpu": 34.75, + "tokens/trainable": 171646 + }, + { + "epoch": 1.46875, + "grad_norm": 0.008862881921231747, + "learning_rate": 2.645931522709877e-05, + "loss": 0.00019395005074329674, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 376, + "tokens/total": 11382528, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 172063 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.03388524800539017, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0004883318324573338, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00049, + "step": 377, + "tokens/total": 11412832, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 172539 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.02546733431518078, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005159692373126745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11442864, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 172997 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.045234713703393936, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0004851966805290431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 379, + "tokens/total": 11473328, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 173481 + }, + { + "epoch": 1.484375, + "grad_norm": 0.011559339240193367, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00026303555932827294, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00026, + "step": 380, + "tokens/total": 11503504, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 173946 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.01947280764579773, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0004145601997151971, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00041, + "step": 381, + "tokens/total": 11533920, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 174417 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.03896784782409668, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.0005209866212680936, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00052, + "step": 382, + "tokens/total": 11564128, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 174872 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.012542990036308765, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00026536238146945834, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00027, + "step": 383, + "tokens/total": 11594672, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 175313 + }, + { + "epoch": 1.5, + "grad_norm": 0.23138198256492615, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.011072758585214615, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01113, + "step": 384, + "tokens/total": 11625056, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 175750 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.5275915861129761, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.016837235540151596, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01698, + "step": 385, + "tokens/total": 11655488, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 176232 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.008643269538879395, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00011418825306463987, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 386, + "tokens/total": 11685968, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 176680 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.00684578949585557, + "learning_rate": 2.406442693028651e-05, + "loss": 0.0001287878112634644, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 387, + "tokens/total": 11716176, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 177144 + }, + { + "epoch": 1.515625, + "grad_norm": 0.07777013629674911, + "learning_rate": 2.3854255584458547e-05, + "loss": 0.001054117688909173, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00105, + "step": 388, + "tokens/total": 11746704, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 177630 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.021813517436385155, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.00045925029553472996, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00046, + "step": 389, + "tokens/total": 11777008, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 178079 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2766229808330536, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0063839266076684, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0064, + "step": 390, + "tokens/total": 11807168, + "tokens/train_per_sec_per_gpu": 36.99, + "tokens/trainable": 178552 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.5076630711555481, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.003972330130636692, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00398, + "step": 391, + "tokens/total": 11837776, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 179021 + }, + { + "epoch": 1.53125, + "grad_norm": 0.046427834779024124, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00048776037874631584, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 392, + "tokens/total": 11867968, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 179478 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.12148728966712952, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00316976523026824, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00317, + "step": 393, + "tokens/total": 11898320, + "tokens/train_per_sec_per_gpu": 33.27, + "tokens/trainable": 179910 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02432228811085224, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00029762828489765525, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0003, + "step": 394, + "tokens/total": 11928624, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 180361 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.01567767933011055, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.00036671021371148527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00037, + "step": 395, + "tokens/total": 11959088, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 180774 + }, + { + "epoch": 1.546875, + "grad_norm": 0.02068573608994484, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00030358770163729787, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0003, + "step": 396, + "tokens/total": 11989504, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 181246 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.2614375948905945, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00875169225037098, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00879, + "step": 397, + "tokens/total": 12019904, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 181734 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.037434663623571396, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00044167632586322725, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00044, + "step": 398, + "tokens/total": 12050384, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 182169 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.16984692215919495, + "learning_rate": 2.1629798596457056e-05, + "loss": 0.0007633643108420074, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00076, + "step": 399, + "tokens/total": 12080576, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 182641 + }, + { + "epoch": 1.5625, + "grad_norm": 0.17288610339164734, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.0036359927617013454, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00364, + "step": 400, + "tokens/total": 12110896, + "tokens/train_per_sec_per_gpu": 26.93, + "tokens/trainable": 183056 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.056392524391412735, + "learning_rate": 2.124308220868431e-05, + "loss": 0.0008229271625168622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00082, + "step": 401, + "tokens/total": 12140784, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 183504 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0365372970700264, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0007215660298243165, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00072, + "step": 402, + "tokens/total": 12171024, + "tokens/train_per_sec_per_gpu": 39.29, + "tokens/trainable": 184004 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.016568642109632492, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.00029834112501703203, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0003, + "step": 403, + "tokens/total": 12201184, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 184476 + }, + { + "epoch": 1.578125, + "grad_norm": 0.06592518836259842, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0007513194577768445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 404, + "tokens/total": 12231232, + "tokens/train_per_sec_per_gpu": 32.35, + "tokens/trainable": 184905 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.013345538638532162, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001536268973723054, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00015, + "step": 405, + "tokens/total": 12261280, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 185338 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.11538074165582657, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.001165087684057653, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00117, + "step": 406, + "tokens/total": 12291520, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 185792 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.01569611392915249, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00031081197084859014, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00031, + "step": 407, + "tokens/total": 12321824, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 186251 + }, + { + "epoch": 1.59375, + "grad_norm": 0.02253701724112034, + "learning_rate": 1.993423839463052e-05, + "loss": 0.00034153330489061773, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00034, + "step": 408, + "tokens/total": 12352176, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 186689 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.40578708052635193, + "learning_rate": 1.975303621298445e-05, + "loss": 0.005957326851785183, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00598, + "step": 409, + "tokens/total": 12382608, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 187121 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.15000925958156586, + "learning_rate": 1.957330080137385e-05, + "loss": 0.0013339307624846697, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00133, + "step": 410, + "tokens/total": 12413152, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 187568 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.006941859144717455, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00013002814375795424, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 411, + "tokens/total": 12443600, + "tokens/train_per_sec_per_gpu": 34.76, + "tokens/trainable": 188010 + }, + { + "epoch": 1.609375, + "grad_norm": 0.12384258210659027, + "learning_rate": 1.9218260145006073e-05, + "loss": 0.0010012347484007478, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.001, + "step": 412, + "tokens/total": 12471952, + "tokens/train_per_sec_per_gpu": 39.3, + "tokens/trainable": 188446 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.3418160378932953, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00740136718377471, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00743, + "step": 413, + "tokens/total": 12502480, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 188897 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.006758023519068956, + "learning_rate": 1.8869175523676064e-05, + "loss": 6.429506174754351e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00006, + "step": 414, + "tokens/total": 12532720, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 189366 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.19964486360549927, + "learning_rate": 1.869688492349885e-05, + "loss": 0.005143567454069853, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00516, + "step": 415, + "tokens/total": 12562848, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 189779 + }, + { + "epoch": 1.625, + "grad_norm": 0.16783277690410614, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007222781423479319, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 416, + "tokens/total": 12593328, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 190245 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.018791699782013893, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.0002913455246016383, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00029, + "step": 417, + "tokens/total": 12623760, + "tokens/train_per_sec_per_gpu": 35.59, + "tokens/trainable": 190737 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.09821721911430359, + "learning_rate": 1.8189105812005714e-05, + "loss": 0.0005217839498072863, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00052, + "step": 418, + "tokens/total": 12654272, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 191155 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.071963369846344, + "learning_rate": 1.802290048317732e-05, + "loss": 0.0007684896700084209, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00077, + "step": 419, + "tokens/total": 12684544, + "tokens/train_per_sec_per_gpu": 31.98, + "tokens/trainable": 191577 + }, + { + "epoch": 1.640625, + "grad_norm": 0.041365377604961395, + "learning_rate": 1.785823392239424e-05, + "loss": 0.0007164644775912166, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00072, + "step": 420, + "tokens/total": 12715008, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 192051 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.01246409397572279, + "learning_rate": 1.7695112982104225e-05, + "loss": 0.0002488879836164415, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00025, + "step": 421, + "tokens/total": 12745232, + "tokens/train_per_sec_per_gpu": 31.77, + "tokens/trainable": 192500 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.00797255989164114, + "learning_rate": 1.7533544450435433e-05, + "loss": 8.724055078346282e-05, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 422, + "tokens/total": 12775824, + "tokens/train_per_sec_per_gpu": 33.23, + "tokens/trainable": 192955 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.029961727559566498, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.00036373038892634213, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00036, + "step": 423, + "tokens/total": 12806240, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 193462 + }, + { + "epoch": 1.65625, + "grad_norm": 0.11438705027103424, + "learning_rate": 1.721509144218405e-05, + "loss": 0.0008654020493850112, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00087, + "step": 424, + "tokens/total": 12836512, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 193943 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.11104747653007507, + "learning_rate": 1.705822021773101e-05, + "loss": 0.0019330887589603662, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00193, + "step": 425, + "tokens/total": 12866704, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 194360 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.005441099405288696, + "learning_rate": 1.69029279056068e-05, + "loss": 0.00012030347716063261, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00012, + "step": 426, + "tokens/total": 12896912, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 194848 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.010912942700088024, + "learning_rate": 1.6749220968158415e-05, + "loss": 0.00012806981976609677, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00013, + "step": 427, + "tokens/total": 12927248, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 195314 + }, + { + "epoch": 1.671875, + "grad_norm": 0.007124146446585655, + "learning_rate": 1.659710580175893e-05, + "loss": 9.905424667522311e-05, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0001, + "step": 428, + "tokens/total": 12957632, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 195732 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.23617680370807648, + "learning_rate": 1.644658873654133e-05, + "loss": 0.0024689119309186935, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00247, + "step": 429, + "tokens/total": 12987824, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 196197 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.29176223278045654, + "learning_rate": 1.629767603613508e-05, + "loss": 0.005283118691295385, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0053, + "step": 430, + "tokens/total": 13018208, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 196678 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.030829036608338356, + "learning_rate": 1.615037389740547e-05, + "loss": 0.0004377455043140799, + "memory/device_reserved (GiB)": 36.36, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00044, + "step": 431, + "tokens/total": 13048800, + "tokens/train_per_sec_per_gpu": 31.52, + "tokens/trainable": 197109 + }, + { + "epoch": 1.6875, + "grad_norm": 0.023997988551855087, + "learning_rate": 1.600468845019576e-05, + "loss": 0.0002723708748817444, + "memory/device_reserved (GiB)": 36.36, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00027, + "step": 432, + "tokens/total": 13079392, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 197607 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.03983060643076897, + "learning_rate": 1.5860625757072092e-05, + "loss": 0.0006240076618269086, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00062, + "step": 433, + "tokens/total": 13109920, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 198066 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.03856171295046806, + "learning_rate": 1.571819181307116e-05, + "loss": 0.0005622098105959594, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00056, + "step": 434, + "tokens/total": 13140064, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 198484 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.07453715056180954, + "learning_rate": 1.557739254545075e-05, + "loss": 0.0008548495243303478, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00086, + "step": 435, + "tokens/total": 13170656, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 198957 + }, + { + "epoch": 1.703125, + "grad_norm": 0.0035155205987393856, + "learning_rate": 1.543823381344311e-05, + "loss": 5.2120005420874804e-05, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 436, + "tokens/total": 13201056, + "tokens/train_per_sec_per_gpu": 33.78, + "tokens/trainable": 199405 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.9568848013877869, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.00043475639540702105, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00043, + "step": 437, + "tokens/total": 13231584, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 199888 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.010019529610872269, + "learning_rate": 1.5164861051607254e-05, + "loss": 0.000111720735731069, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 438, + "tokens/total": 13261888, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 200396 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.008281780406832695, + "learning_rate": 1.5030658397935521e-05, + "loss": 0.00017071102047339082, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 439, + "tokens/total": 13291936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 200843 + }, + { + "epoch": 1.71875, + "grad_norm": 0.06077976152300835, + "learning_rate": 1.4898119031716104e-05, + "loss": 0.0007504730019718409, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 440, + "tokens/total": 13322224, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 201315 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.12958590686321259, + "learning_rate": 1.476724846845306e-05, + "loss": 0.001083776238374412, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00108, + "step": 441, + "tokens/total": 13352640, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 201782 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.04559873417019844, + "learning_rate": 1.463805215420471e-05, + "loss": 0.0005228023510426283, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00052, + "step": 442, + "tokens/total": 13382560, + "tokens/train_per_sec_per_gpu": 31.27, + "tokens/trainable": 202222 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.02645108290016651, + "learning_rate": 1.451053546535705e-05, + "loss": 0.0003615481255110353, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00036, + "step": 443, + "tokens/total": 13412976, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 202660 + }, + { + "epoch": 1.734375, + "grad_norm": 0.021894006058573723, + "learning_rate": 1.438470370840001e-05, + "loss": 0.000229351528105326, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00023, + "step": 444, + "tokens/total": 13442720, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 203068 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.1398804783821106, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.0013714064843952656, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00137, + "step": 445, + "tokens/total": 13472928, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 203527 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.024557381868362427, + "learning_rate": 1.413811586531508e-05, + "loss": 0.00023480730305891484, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00023, + "step": 446, + "tokens/total": 13502992, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 203994 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.021804476156830788, + "learning_rate": 1.4017370040713884e-05, + "loss": 0.00022578673087991774, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00023, + "step": 447, + "tokens/total": 13533184, + "tokens/train_per_sec_per_gpu": 38.06, + "tokens/trainable": 204430 + }, + { + "epoch": 1.75, + "grad_norm": 0.23265990614891052, + "learning_rate": 1.3898329670629645e-05, + "loss": 0.002107386477291584, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00211, + "step": 448, + "tokens/total": 13563648, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 204886 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.015779122710227966, + "learning_rate": 1.3780999708818058e-05, + "loss": 0.00014222922618500888, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00014, + "step": 449, + "tokens/total": 13594032, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 205356 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.09757973998785019, + "learning_rate": 1.3665385037857758e-05, + "loss": 0.0012260058429092169, + "memory/device_reserved (GiB)": 35.69, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00123, + "step": 450, + "tokens/total": 13624368, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 205782 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.05971763655543327, + "learning_rate": 1.3551490468947126e-05, + "loss": 0.0006241232040338218, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00062, + "step": 451, + "tokens/total": 13654768, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 206219 + }, + { + "epoch": 1.765625, + "grad_norm": 0.009018263779580593, + "learning_rate": 1.3439320741704075e-05, + "loss": 5.874139242223464e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 452, + "tokens/total": 13684960, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 206677 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.011312981136143208, + "learning_rate": 1.3328880523968808e-05, + "loss": 8.52083321660757e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 453, + "tokens/total": 13715488, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 207160 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.0022447824012488127, + "learning_rate": 1.3220174411609587e-05, + "loss": 3.376982203917578e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 454, + "tokens/total": 13745856, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 207605 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.017390765249729156, + "learning_rate": 1.3113206928331471e-05, + "loss": 0.00019571901066228747, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0002, + "step": 455, + "tokens/total": 13776192, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 208047 + }, + { + "epoch": 1.78125, + "grad_norm": 0.048852503299713135, + "learning_rate": 1.300798252548806e-05, + "loss": 0.0005775241879746318, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00058, + "step": 456, + "tokens/total": 13806496, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 208476 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.3256935179233551, + "learning_rate": 1.2904505581896265e-05, + "loss": 0.0016588940052315593, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00166, + "step": 457, + "tokens/total": 13836768, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 208970 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.06032567098736763, + "learning_rate": 1.2802780403654082e-05, + "loss": 0.000697342911735177, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0007, + "step": 458, + "tokens/total": 13867152, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 209456 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.007340370211750269, + "learning_rate": 1.2702811223961408e-05, + "loss": 3.1743944418849424e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00003, + "step": 459, + "tokens/total": 13897360, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 209907 + }, + { + "epoch": 1.796875, + "grad_norm": 0.16073353588581085, + "learning_rate": 1.2604602202943861e-05, + "loss": 0.001599560840986669, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0016, + "step": 460, + "tokens/total": 13927792, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 210366 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.5971834659576416, + "learning_rate": 1.2508157427479686e-05, + "loss": 0.002240244299173355, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.65, + "memory/max_allocated (GiB)": 33.65, + "ppl": 1.00224, + "step": 461, + "tokens/total": 13957856, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 210835 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.15249626338481903, + "learning_rate": 1.2413480911029655e-05, + "loss": 0.0012240726500749588, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00122, + "step": 462, + "tokens/total": 13986080, + "tokens/train_per_sec_per_gpu": 30.91, + "tokens/trainable": 211259 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.005158649291843176, + "learning_rate": 1.2320576593470082e-05, + "loss": 3.6364894185680896e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00004, + "step": 463, + "tokens/total": 14016288, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 211721 + }, + { + "epoch": 1.8125, + "grad_norm": 0.11264989525079727, + "learning_rate": 1.2229448340928828e-05, + "loss": 0.001997420797124505, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.002, + "step": 464, + "tokens/total": 14046288, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 212159 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.020843395963311195, + "learning_rate": 1.2140099945624458e-05, + "loss": 0.0001486139663029462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00015, + "step": 465, + "tokens/total": 14076512, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 212644 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.0014335883315652609, + "learning_rate": 1.205253512570841e-05, + "loss": 2.982833029818721e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 466, + "tokens/total": 14106608, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 213084 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.17584311962127686, + "learning_rate": 1.1966757525110255e-05, + "loss": 0.0020335109438747168, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00204, + "step": 467, + "tokens/total": 14136864, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 213523 + }, + { + "epoch": 1.828125, + "grad_norm": 0.012916951440274715, + "learning_rate": 1.1882770713386095e-05, + "loss": 0.00013626206782646477, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00014, + "step": 468, + "tokens/total": 14166896, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 213963 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.13297978043556213, + "learning_rate": 1.180057818556998e-05, + "loss": 0.0010875939624384046, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00109, + "step": 469, + "tokens/total": 14197568, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 214431 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.0005420309607870877, + "learning_rate": 1.1720183362028494e-05, + "loss": 1.2743539627990685e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00001, + "step": 470, + "tokens/total": 14225984, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 214865 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.3001365065574646, + "learning_rate": 1.1641589588318387e-05, + "loss": 0.0015361867845058441, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00154, + "step": 471, + "tokens/total": 14256512, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 215358 + }, + { + "epoch": 1.84375, + "grad_norm": 0.014027237892150879, + "learning_rate": 1.1564800135047418e-05, + "loss": 0.00015228531265165657, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00015, + "step": 472, + "tokens/total": 14286640, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 215800 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.08512959629297256, + "learning_rate": 1.148981819773816e-05, + "loss": 0.000548110285308212, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00055, + "step": 473, + "tokens/total": 14317024, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 216245 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.004200903698801994, + "learning_rate": 1.1416646896695086e-05, + "loss": 4.4981639803154394e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00004, + "step": 474, + "tokens/total": 14347088, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 216685 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.4064914584159851, + "learning_rate": 1.1345289276874717e-05, + "loss": 0.002648166147992015, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00265, + "step": 475, + "tokens/total": 14377152, + "tokens/train_per_sec_per_gpu": 31.31, + "tokens/trainable": 217092 + }, + { + "epoch": 1.859375, + "grad_norm": 0.0013920688070356846, + "learning_rate": 1.1275748307758873e-05, + "loss": 1.866836828412488e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 476, + "tokens/total": 14407568, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 217582 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.0018797408556565642, + "learning_rate": 1.1208026883231147e-05, + "loss": 2.958982076961547e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00003, + "step": 477, + "tokens/total": 14437792, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 218049 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.10846851021051407, + "learning_rate": 1.1142127821456433e-05, + "loss": 0.000766873883549124, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00077, + "step": 478, + "tokens/total": 14468400, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 218522 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.08651807904243469, + "learning_rate": 1.1078053864763674e-05, + "loss": 0.0007375412969850004, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00074, + "step": 479, + "tokens/total": 14498416, + "tokens/train_per_sec_per_gpu": 33.32, + "tokens/trainable": 218970 + }, + { + "epoch": 1.875, + "grad_norm": 0.19086171686649323, + "learning_rate": 1.1015807679531756e-05, + "loss": 0.002163141267374158, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00217, + "step": 480, + "tokens/total": 14528976, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 219411 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9.861665047363261e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-480/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..5bc2f0988df480edd572b506b460277309280203 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6f7da5abb2cf5cc22da25374898331b609fa27f726a3db2faafa9214b3a4ee3d +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..3ffa21572a31c75b79b746a8b6e5b62713b4a98e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4af1671343e680b761c384855b03b2f3edba0c1c73e238f273b4ec981203dd8 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..451958613dee729cb814f0dd672153d0f855dbd2 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ad796432b92ca02dee812ca2f0084633293af1832f48212de91b45c8633bc0c9 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..22325b3b3d08e0697784fe21b8c321485a51f7b5 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:697fe8894f7795467da8b1a7ccebf8570b28357dc457e2ba8ccdb525507cfef4 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..748b8a5a3cc64401e49842ec7c40e08259a929dc --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/tokens_state.json @@ -0,0 +1 @@ +{"total": 15498240, "trainable": 234124} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..be35849dfdd0303c1d2623f61e21f66942c47521 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/trainer_state.json @@ -0,0 +1,7202 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 2.0, + "eval_steps": 500, + "global_step": 512, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + }, + { + "epoch": 0.37890625, + "grad_norm": 0.1258164346218109, + "learning_rate": 9.53619478457953e-05, + "loss": 0.002208392135798931, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00221, + "step": 97, + "tokens/total": 2936560, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 44300 + }, + { + "epoch": 0.3828125, + "grad_norm": 0.43118366599082947, + "learning_rate": 9.523275153154695e-05, + "loss": 0.015600357204675674, + "memory/device_reserved (GiB)": 35.19, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01572, + "step": 98, + "tokens/total": 2966880, + "tokens/train_per_sec_per_gpu": 39.37, + "tokens/trainable": 44791 + }, + { + "epoch": 0.38671875, + "grad_norm": 0.505718469619751, + "learning_rate": 9.51018809682839e-05, + "loss": 0.011874521151185036, + "memory/device_reserved (GiB)": 35.76, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.01195, + "step": 99, + "tokens/total": 2997472, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 45254 + }, + { + "epoch": 0.390625, + "grad_norm": 3.060197591781616, + "learning_rate": 9.49693416020645e-05, + "loss": 0.014575858600437641, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01468, + "step": 100, + "tokens/total": 3027840, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 45698 + }, + { + "epoch": 0.39453125, + "grad_norm": 0.38254642486572266, + "learning_rate": 9.483513894839276e-05, + "loss": 0.009431221522390842, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00948, + "step": 101, + "tokens/total": 3058080, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 46141 + }, + { + "epoch": 0.3984375, + "grad_norm": 0.8782851099967957, + "learning_rate": 9.469927859198888e-05, + "loss": 0.05128197371959686, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.05262, + "step": 102, + "tokens/total": 3088656, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 46598 + }, + { + "epoch": 0.40234375, + "grad_norm": 0.42221176624298096, + "learning_rate": 9.456176618655689e-05, + "loss": 0.009157870896160603, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0092, + "step": 103, + "tokens/total": 3119232, + "tokens/train_per_sec_per_gpu": 32.45, + "tokens/trainable": 47065 + }, + { + "epoch": 0.40625, + "grad_norm": 1.2625210285186768, + "learning_rate": 9.442260745454927e-05, + "loss": 0.01384878158569336, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01395, + "step": 104, + "tokens/total": 3149504, + "tokens/train_per_sec_per_gpu": 38.29, + "tokens/trainable": 47554 + }, + { + "epoch": 0.41015625, + "grad_norm": 0.3072832226753235, + "learning_rate": 9.428180818692884e-05, + "loss": 0.00823851116001606, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00827, + "step": 105, + "tokens/total": 3179872, + "tokens/train_per_sec_per_gpu": 36.88, + "tokens/trainable": 48047 + }, + { + "epoch": 0.4140625, + "grad_norm": 0.7415916323661804, + "learning_rate": 9.413937424292791e-05, + "loss": 0.03210819885134697, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.03263, + "step": 106, + "tokens/total": 3208128, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 48498 + }, + { + "epoch": 0.41796875, + "grad_norm": 0.25024500489234924, + "learning_rate": 9.399531154980424e-05, + "loss": 0.00918180774897337, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00922, + "step": 107, + "tokens/total": 3238224, + "tokens/train_per_sec_per_gpu": 31.18, + "tokens/trainable": 48933 + }, + { + "epoch": 0.421875, + "grad_norm": 0.5457918047904968, + "learning_rate": 9.384962610259455e-05, + "loss": 0.015052877366542816, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01517, + "step": 108, + "tokens/total": 3268592, + "tokens/train_per_sec_per_gpu": 38.48, + "tokens/trainable": 49404 + }, + { + "epoch": 0.42578125, + "grad_norm": 0.2508637607097626, + "learning_rate": 9.370232396386494e-05, + "loss": 0.007351120002567768, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00738, + "step": 109, + "tokens/total": 3298720, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 49871 + }, + { + "epoch": 0.4296875, + "grad_norm": 0.2964157462120056, + "learning_rate": 9.355341126345868e-05, + "loss": 0.014206478372216225, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.01431, + "step": 110, + "tokens/total": 3329040, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 50311 + }, + { + "epoch": 0.43359375, + "grad_norm": 0.3096546232700348, + "learning_rate": 9.340289419824107e-05, + "loss": 0.010700306855142117, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01076, + "step": 111, + "tokens/total": 3359248, + "tokens/train_per_sec_per_gpu": 33.86, + "tokens/trainable": 50754 + }, + { + "epoch": 0.4375, + "grad_norm": 0.4973372220993042, + "learning_rate": 9.325077903184159e-05, + "loss": 0.013185497373342514, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01327, + "step": 112, + "tokens/total": 3389280, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 51200 + }, + { + "epoch": 0.44140625, + "grad_norm": 0.4105257987976074, + "learning_rate": 9.30970720943932e-05, + "loss": 0.0069708251394331455, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.007, + "step": 113, + "tokens/total": 3419728, + "tokens/train_per_sec_per_gpu": 33.22, + "tokens/trainable": 51656 + }, + { + "epoch": 0.4453125, + "grad_norm": 0.2707289159297943, + "learning_rate": 9.2941779782269e-05, + "loss": 0.0033873419743031263, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00339, + "step": 114, + "tokens/total": 3449984, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 52069 + }, + { + "epoch": 0.44921875, + "grad_norm": 0.5602349042892456, + "learning_rate": 9.278490855781596e-05, + "loss": 0.00848664902150631, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00852, + "step": 115, + "tokens/total": 3480352, + "tokens/train_per_sec_per_gpu": 36.94, + "tokens/trainable": 52559 + }, + { + "epoch": 0.453125, + "grad_norm": 0.5224027037620544, + "learning_rate": 9.262646494908604e-05, + "loss": 0.02242768369615078, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02268, + "step": 116, + "tokens/total": 3510656, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 53061 + }, + { + "epoch": 0.45703125, + "grad_norm": 1.2213423252105713, + "learning_rate": 9.246645554956457e-05, + "loss": 0.020464815199375153, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.02068, + "step": 117, + "tokens/total": 3541104, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 53520 + }, + { + "epoch": 0.4609375, + "grad_norm": 1.30115807056427, + "learning_rate": 9.230488701789578e-05, + "loss": 0.025597743690013885, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02593, + "step": 118, + "tokens/total": 3571744, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 53976 + }, + { + "epoch": 0.46484375, + "grad_norm": 1.0988223552703857, + "learning_rate": 9.214176607760577e-05, + "loss": 0.02888210117816925, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.0293, + "step": 119, + "tokens/total": 3601808, + "tokens/train_per_sec_per_gpu": 34.93, + "tokens/trainable": 54408 + }, + { + "epoch": 0.46875, + "grad_norm": 0.3977462649345398, + "learning_rate": 9.197709951682268e-05, + "loss": 0.01678757183253765, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01693, + "step": 120, + "tokens/total": 3632096, + "tokens/train_per_sec_per_gpu": 37.14, + "tokens/trainable": 54884 + }, + { + "epoch": 0.47265625, + "grad_norm": 0.4872536063194275, + "learning_rate": 9.181089418799428e-05, + "loss": 0.022638533264398575, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0229, + "step": 121, + "tokens/total": 3662416, + "tokens/train_per_sec_per_gpu": 33.15, + "tokens/trainable": 55303 + }, + { + "epoch": 0.4765625, + "grad_norm": 0.4663882255554199, + "learning_rate": 9.164315700760271e-05, + "loss": 0.023266036063432693, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02354, + "step": 122, + "tokens/total": 3692992, + "tokens/train_per_sec_per_gpu": 31.94, + "tokens/trainable": 55743 + }, + { + "epoch": 0.48046875, + "grad_norm": 0.230746328830719, + "learning_rate": 9.147389495587671e-05, + "loss": 0.007810275070369244, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00784, + "step": 123, + "tokens/total": 3722992, + "tokens/train_per_sec_per_gpu": 33.95, + "tokens/trainable": 56187 + }, + { + "epoch": 0.484375, + "grad_norm": 0.17318949103355408, + "learning_rate": 9.130311507650116e-05, + "loss": 0.006973203271627426, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.007, + "step": 124, + "tokens/total": 3753184, + "tokens/train_per_sec_per_gpu": 33.63, + "tokens/trainable": 56632 + }, + { + "epoch": 0.48828125, + "grad_norm": 0.19994892179965973, + "learning_rate": 9.113082447632394e-05, + "loss": 0.0068689389154314995, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00689, + "step": 125, + "tokens/total": 3783440, + "tokens/train_per_sec_per_gpu": 37.26, + "tokens/trainable": 57124 + }, + { + "epoch": 0.4921875, + "grad_norm": 0.11329996585845947, + "learning_rate": 9.09570303250602e-05, + "loss": 0.004266452044248581, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00428, + "step": 126, + "tokens/total": 3813840, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 57570 + }, + { + "epoch": 0.49609375, + "grad_norm": 0.4156385362148285, + "learning_rate": 9.078173985499394e-05, + "loss": 0.02041183039546013, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02062, + "step": 127, + "tokens/total": 3844464, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 58048 + }, + { + "epoch": 0.5, + "grad_norm": 0.09819997847080231, + "learning_rate": 9.060496036067713e-05, + "loss": 0.0031540761701762676, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00316, + "step": 128, + "tokens/total": 3872816, + "tokens/train_per_sec_per_gpu": 35.79, + "tokens/trainable": 58497 + }, + { + "epoch": 0.50390625, + "grad_norm": 0.34266355633735657, + "learning_rate": 9.042669919862615e-05, + "loss": 0.017772674560546875, + "memory/device_reserved (GiB)": 36.0, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01793, + "step": 129, + "tokens/total": 3902992, + "tokens/train_per_sec_per_gpu": 33.44, + "tokens/trainable": 58947 + }, + { + "epoch": 0.5078125, + "grad_norm": 0.3253064751625061, + "learning_rate": 9.024696378701557e-05, + "loss": 0.018172409385442734, + "memory/device_reserved (GiB)": 34.74, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01834, + "step": 130, + "tokens/total": 3933456, + "tokens/train_per_sec_per_gpu": 37.56, + "tokens/trainable": 59415 + }, + { + "epoch": 0.51171875, + "grad_norm": 0.3167641758918762, + "learning_rate": 9.006576160536948e-05, + "loss": 0.008536357432603836, + "memory/device_reserved (GiB)": 34.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00857, + "step": 131, + "tokens/total": 3963536, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 59851 + }, + { + "epoch": 0.515625, + "grad_norm": 0.12403249740600586, + "learning_rate": 8.988310019425035e-05, + "loss": 0.0019342785235494375, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00194, + "step": 132, + "tokens/total": 3994000, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 60325 + }, + { + "epoch": 0.51953125, + "grad_norm": 0.25862130522727966, + "learning_rate": 8.969898715494506e-05, + "loss": 0.01549853477627039, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01562, + "step": 133, + "tokens/total": 4024352, + "tokens/train_per_sec_per_gpu": 35.97, + "tokens/trainable": 60818 + }, + { + "epoch": 0.5234375, + "grad_norm": 0.3527744710445404, + "learning_rate": 8.951343014914869e-05, + "loss": 0.006827985402196646, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.39, + "memory/max_allocated (GiB)": 33.39, + "ppl": 1.00685, + "step": 134, + "tokens/total": 4052832, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 61263 + }, + { + "epoch": 0.52734375, + "grad_norm": 0.3797135353088379, + "learning_rate": 8.932643689864568e-05, + "loss": 0.010647830553352833, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.0107, + "step": 135, + "tokens/total": 4083216, + "tokens/train_per_sec_per_gpu": 36.58, + "tokens/trainable": 61734 + }, + { + "epoch": 0.53125, + "grad_norm": 0.3589562177658081, + "learning_rate": 8.913801518498845e-05, + "loss": 0.022732166573405266, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.02299, + "step": 136, + "tokens/total": 4113392, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 62210 + }, + { + "epoch": 0.53515625, + "grad_norm": 0.34385553002357483, + "learning_rate": 8.894817284917364e-05, + "loss": 0.017764274030923843, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01792, + "step": 137, + "tokens/total": 4143696, + "tokens/train_per_sec_per_gpu": 36.87, + "tokens/trainable": 62707 + }, + { + "epoch": 0.5390625, + "grad_norm": 0.24294431507587433, + "learning_rate": 8.875691779131569e-05, + "loss": 0.01984233781695366, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02004, + "step": 138, + "tokens/total": 4174000, + "tokens/train_per_sec_per_gpu": 37.17, + "tokens/trainable": 63166 + }, + { + "epoch": 0.54296875, + "grad_norm": 0.6662021279335022, + "learning_rate": 8.856425797031829e-05, + "loss": 0.012637479230761528, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01272, + "step": 139, + "tokens/total": 4204432, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 63646 + }, + { + "epoch": 0.546875, + "grad_norm": 0.26386311650276184, + "learning_rate": 8.837020140354295e-05, + "loss": 0.014607764780521393, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.42, + "memory/max_allocated (GiB)": 33.42, + "ppl": 1.01471, + "step": 140, + "tokens/total": 4232752, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 64063 + }, + { + "epoch": 0.55078125, + "grad_norm": 0.2475082129240036, + "learning_rate": 8.817475616647554e-05, + "loss": 0.012658985331654549, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01274, + "step": 141, + "tokens/total": 4262976, + "tokens/train_per_sec_per_gpu": 35.67, + "tokens/trainable": 64520 + }, + { + "epoch": 0.5546875, + "grad_norm": 0.21145126223564148, + "learning_rate": 8.797793039239017e-05, + "loss": 0.009737009182572365, + "memory/device_reserved (GiB)": 35.79, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00978, + "step": 142, + "tokens/total": 4293520, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 64958 + }, + { + "epoch": 0.55859375, + "grad_norm": 0.2296362668275833, + "learning_rate": 8.777973227201069e-05, + "loss": 0.013068155385553837, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01315, + "step": 143, + "tokens/total": 4323520, + "tokens/train_per_sec_per_gpu": 38.3, + "tokens/trainable": 65437 + }, + { + "epoch": 0.5625, + "grad_norm": 0.2896386981010437, + "learning_rate": 8.758017005316988e-05, + "loss": 0.010017371736466885, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01007, + "step": 144, + "tokens/total": 4353616, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 65903 + }, + { + "epoch": 0.56640625, + "grad_norm": 0.41864296793937683, + "learning_rate": 8.737925204046629e-05, + "loss": 0.025722038000822067, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.02606, + "step": 145, + "tokens/total": 4383856, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 66334 + }, + { + "epoch": 0.5703125, + "grad_norm": 0.27801141142845154, + "learning_rate": 8.717698659491851e-05, + "loss": 0.017308663576841354, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01746, + "step": 146, + "tokens/total": 4414128, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 66785 + }, + { + "epoch": 0.57421875, + "grad_norm": 0.575922966003418, + "learning_rate": 8.697338213361735e-05, + "loss": 0.017252381891012192, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0174, + "step": 147, + "tokens/total": 4444384, + "tokens/train_per_sec_per_gpu": 35.81, + "tokens/trainable": 67199 + }, + { + "epoch": 0.578125, + "grad_norm": 0.4430011510848999, + "learning_rate": 8.676844712937552e-05, + "loss": 0.022439993917942047, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02269, + "step": 148, + "tokens/total": 4474912, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 67683 + }, + { + "epoch": 0.58203125, + "grad_norm": 0.26630517840385437, + "learning_rate": 8.656219011037509e-05, + "loss": 0.010438371449708939, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01049, + "step": 149, + "tokens/total": 4505328, + "tokens/train_per_sec_per_gpu": 34.4, + "tokens/trainable": 68127 + }, + { + "epoch": 0.5859375, + "grad_norm": 0.15251637995243073, + "learning_rate": 8.63546196598125e-05, + "loss": 0.004219424910843372, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00423, + "step": 150, + "tokens/total": 4535424, + "tokens/train_per_sec_per_gpu": 33.75, + "tokens/trainable": 68595 + }, + { + "epoch": 0.58984375, + "grad_norm": 0.21238818764686584, + "learning_rate": 8.614574441554145e-05, + "loss": 0.01187204197049141, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01194, + "step": 151, + "tokens/total": 4565536, + "tokens/train_per_sec_per_gpu": 30.68, + "tokens/trainable": 69062 + }, + { + "epoch": 0.59375, + "grad_norm": 0.5052844285964966, + "learning_rate": 8.593557306971349e-05, + "loss": 0.007624611258506775, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00765, + "step": 152, + "tokens/total": 4596048, + "tokens/train_per_sec_per_gpu": 35.96, + "tokens/trainable": 69509 + }, + { + "epoch": 0.59765625, + "grad_norm": 0.46278703212738037, + "learning_rate": 8.572411436841618e-05, + "loss": 0.01224217563867569, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01232, + "step": 153, + "tokens/total": 4626480, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 69972 + }, + { + "epoch": 0.6015625, + "grad_norm": 0.4227867126464844, + "learning_rate": 8.551137711130922e-05, + "loss": 0.0060340641066432, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00605, + "step": 154, + "tokens/total": 4656800, + "tokens/train_per_sec_per_gpu": 37.3, + "tokens/trainable": 70439 + }, + { + "epoch": 0.60546875, + "grad_norm": 0.5990466475486755, + "learning_rate": 8.529737015125824e-05, + "loss": 0.018272625282406807, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01844, + "step": 155, + "tokens/total": 4685120, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 70881 + }, + { + "epoch": 0.609375, + "grad_norm": 0.6471202969551086, + "learning_rate": 8.508210239396639e-05, + "loss": 0.01053079217672348, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01059, + "step": 156, + "tokens/total": 4715664, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 71339 + }, + { + "epoch": 0.61328125, + "grad_norm": 0.00898966658860445, + "learning_rate": 8.486558279760375e-05, + "loss": 0.0001828348613344133, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00018, + "step": 157, + "tokens/total": 4745824, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 71801 + }, + { + "epoch": 0.6171875, + "grad_norm": 0.2965949773788452, + "learning_rate": 8.464782037243449e-05, + "loss": 0.014287484809756279, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01439, + "step": 158, + "tokens/total": 4776224, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 72237 + }, + { + "epoch": 0.62109375, + "grad_norm": 0.3070797324180603, + "learning_rate": 8.442882418044202e-05, + "loss": 0.025281894952058792, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.0256, + "step": 159, + "tokens/total": 4806304, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 72664 + }, + { + "epoch": 0.625, + "grad_norm": 0.29432007670402527, + "learning_rate": 8.420860333495179e-05, + "loss": 0.005029833409935236, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00504, + "step": 160, + "tokens/total": 4836688, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 73119 + }, + { + "epoch": 0.62890625, + "grad_norm": 0.38903817534446716, + "learning_rate": 8.398716700025208e-05, + "loss": 0.010762909427285194, + "memory/device_reserved (GiB)": 37.89, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.01082, + "step": 161, + "tokens/total": 4867008, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 73540 + }, + { + "epoch": 0.6328125, + "grad_norm": 0.23661470413208008, + "learning_rate": 8.376452439121266e-05, + "loss": 0.003404806135222316, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00341, + "step": 162, + "tokens/total": 4897584, + "tokens/train_per_sec_per_gpu": 33.18, + "tokens/trainable": 73990 + }, + { + "epoch": 0.63671875, + "grad_norm": 0.23111319541931152, + "learning_rate": 8.354068477290124e-05, + "loss": 0.0172940231859684, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01744, + "step": 163, + "tokens/total": 4927904, + "tokens/train_per_sec_per_gpu": 35.89, + "tokens/trainable": 74454 + }, + { + "epoch": 0.640625, + "grad_norm": 0.7157747745513916, + "learning_rate": 8.331565746019807e-05, + "loss": 0.025878142565488815, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02622, + "step": 164, + "tokens/total": 4958496, + "tokens/train_per_sec_per_gpu": 33.96, + "tokens/trainable": 74905 + }, + { + "epoch": 0.64453125, + "grad_norm": 0.3431304097175598, + "learning_rate": 8.308945181740812e-05, + "loss": 0.003934288397431374, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00394, + "step": 165, + "tokens/total": 4988832, + "tokens/train_per_sec_per_gpu": 31.03, + "tokens/trainable": 75329 + }, + { + "epoch": 0.6484375, + "grad_norm": 0.0704391747713089, + "learning_rate": 8.286207725787153e-05, + "loss": 0.0012416213285177946, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00124, + "step": 166, + "tokens/total": 5019136, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 75770 + }, + { + "epoch": 0.65234375, + "grad_norm": 0.6784317493438721, + "learning_rate": 8.263354324357182e-05, + "loss": 0.008922886103391647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00896, + "step": 167, + "tokens/total": 5049808, + "tokens/train_per_sec_per_gpu": 31.81, + "tokens/trainable": 76231 + }, + { + "epoch": 0.65625, + "grad_norm": 0.41060712933540344, + "learning_rate": 8.240385928474219e-05, + "loss": 0.018168501555919647, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01833, + "step": 168, + "tokens/total": 5080320, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 76682 + }, + { + "epoch": 0.66015625, + "grad_norm": 0.13889877498149872, + "learning_rate": 8.217303493946967e-05, + "loss": 0.0031139992643147707, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00312, + "step": 169, + "tokens/total": 5110656, + "tokens/train_per_sec_per_gpu": 34.87, + "tokens/trainable": 77114 + }, + { + "epoch": 0.6640625, + "grad_norm": 0.6820456385612488, + "learning_rate": 8.194107981329746e-05, + "loss": 0.030903562903404236, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03139, + "step": 170, + "tokens/total": 5141056, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 77563 + }, + { + "epoch": 0.66796875, + "grad_norm": 0.20577676594257355, + "learning_rate": 8.170800355882518e-05, + "loss": 0.0030413041822612286, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00305, + "step": 171, + "tokens/total": 5171312, + "tokens/train_per_sec_per_gpu": 33.7, + "tokens/trainable": 78023 + }, + { + "epoch": 0.671875, + "grad_norm": 0.2194790244102478, + "learning_rate": 8.147381587530713e-05, + "loss": 0.004004347138106823, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00401, + "step": 172, + "tokens/total": 5201744, + "tokens/train_per_sec_per_gpu": 30.23, + "tokens/trainable": 78477 + }, + { + "epoch": 0.67578125, + "grad_norm": 0.5019084811210632, + "learning_rate": 8.123852650824877e-05, + "loss": 0.006000855006277561, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00602, + "step": 173, + "tokens/total": 5232176, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 78924 + }, + { + "epoch": 0.6796875, + "grad_norm": 0.36753812432289124, + "learning_rate": 8.100214524900103e-05, + "loss": 0.004308007657527924, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00432, + "step": 174, + "tokens/total": 5261904, + "tokens/train_per_sec_per_gpu": 30.59, + "tokens/trainable": 79342 + }, + { + "epoch": 0.68359375, + "grad_norm": 0.42971113324165344, + "learning_rate": 8.076468193435301e-05, + "loss": 0.0007982449606060982, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.0008, + "step": 175, + "tokens/total": 5292192, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 79812 + }, + { + "epoch": 0.6875, + "grad_norm": 0.7103844285011292, + "learning_rate": 8.052614644612253e-05, + "loss": 0.005436811596155167, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00545, + "step": 176, + "tokens/total": 5322704, + "tokens/train_per_sec_per_gpu": 36.73, + "tokens/trainable": 80290 + }, + { + "epoch": 0.69140625, + "grad_norm": 0.5654855966567993, + "learning_rate": 8.028654871074489e-05, + "loss": 0.007358669303357601, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00739, + "step": 177, + "tokens/total": 5353040, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 80737 + }, + { + "epoch": 0.6953125, + "grad_norm": 0.035665832459926605, + "learning_rate": 8.004589869885986e-05, + "loss": 0.0001784512132871896, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00018, + "step": 178, + "tokens/total": 5383248, + "tokens/train_per_sec_per_gpu": 35.33, + "tokens/trainable": 81205 + }, + { + "epoch": 0.69921875, + "grad_norm": 0.01852536015212536, + "learning_rate": 7.980420642489674e-05, + "loss": 0.0002031420881394297, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0002, + "step": 179, + "tokens/total": 5413248, + "tokens/train_per_sec_per_gpu": 33.53, + "tokens/trainable": 81652 + }, + { + "epoch": 0.703125, + "grad_norm": 0.10783406347036362, + "learning_rate": 7.95614819466576e-05, + "loss": 0.0007448110263794661, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00075, + "step": 180, + "tokens/total": 5443424, + "tokens/train_per_sec_per_gpu": 35.07, + "tokens/trainable": 82125 + }, + { + "epoch": 0.70703125, + "grad_norm": 0.15577788650989532, + "learning_rate": 7.931773536489872e-05, + "loss": 0.000892514712177217, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00089, + "step": 181, + "tokens/total": 5473680, + "tokens/train_per_sec_per_gpu": 39.35, + "tokens/trainable": 82642 + }, + { + "epoch": 0.7109375, + "grad_norm": 0.7524833679199219, + "learning_rate": 7.907297682291035e-05, + "loss": 0.01774725317955017, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01791, + "step": 182, + "tokens/total": 5504032, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 83123 + }, + { + "epoch": 0.71484375, + "grad_norm": 0.8974100947380066, + "learning_rate": 7.882721650609442e-05, + "loss": 0.016709204763174057, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.01685, + "step": 183, + "tokens/total": 5534224, + "tokens/train_per_sec_per_gpu": 33.26, + "tokens/trainable": 83570 + }, + { + "epoch": 0.71875, + "grad_norm": 0.31876227259635925, + "learning_rate": 7.85804646415409e-05, + "loss": 0.0018143621273338795, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00182, + "step": 184, + "tokens/total": 5564704, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 84076 + }, + { + "epoch": 0.72265625, + "grad_norm": 0.3767867684364319, + "learning_rate": 7.833273149760207e-05, + "loss": 0.030571268871426582, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.03104, + "step": 185, + "tokens/total": 5594912, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 84551 + }, + { + "epoch": 0.7265625, + "grad_norm": 0.025328254327178, + "learning_rate": 7.808402738346527e-05, + "loss": 0.0003344593569636345, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00033, + "step": 186, + "tokens/total": 5625280, + "tokens/train_per_sec_per_gpu": 33.92, + "tokens/trainable": 85012 + }, + { + "epoch": 0.73046875, + "grad_norm": 0.37010785937309265, + "learning_rate": 7.783436264872382e-05, + "loss": 0.009675242006778717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00972, + "step": 187, + "tokens/total": 5655568, + "tokens/train_per_sec_per_gpu": 32.23, + "tokens/trainable": 85452 + }, + { + "epoch": 0.734375, + "grad_norm": 0.32615038752555847, + "learning_rate": 7.758374768294647e-05, + "loss": 0.010340536944568157, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01039, + "step": 188, + "tokens/total": 5685824, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 85942 + }, + { + "epoch": 0.73828125, + "grad_norm": 0.6183643937110901, + "learning_rate": 7.733219291524489e-05, + "loss": 0.00930570624768734, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00935, + "step": 189, + "tokens/total": 5716192, + "tokens/train_per_sec_per_gpu": 36.56, + "tokens/trainable": 86457 + }, + { + "epoch": 0.7421875, + "grad_norm": 0.31889259815216064, + "learning_rate": 7.707970881383977e-05, + "loss": 0.004300425294786692, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00431, + "step": 190, + "tokens/total": 5746784, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 86928 + }, + { + "epoch": 0.74609375, + "grad_norm": 0.34553802013397217, + "learning_rate": 7.682630588562518e-05, + "loss": 0.00664468202739954, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00667, + "step": 191, + "tokens/total": 5777520, + "tokens/train_per_sec_per_gpu": 36.31, + "tokens/trainable": 87384 + }, + { + "epoch": 0.75, + "grad_norm": 0.2645066976547241, + "learning_rate": 7.657199467573129e-05, + "loss": 0.006615218240767717, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00664, + "step": 192, + "tokens/total": 5807952, + "tokens/train_per_sec_per_gpu": 32.03, + "tokens/trainable": 87814 + }, + { + "epoch": 0.75390625, + "grad_norm": 0.3881167471408844, + "learning_rate": 7.631678576708561e-05, + "loss": 0.014919335022568703, + "memory/device_reserved (GiB)": 36.02, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01503, + "step": 193, + "tokens/total": 5838384, + "tokens/train_per_sec_per_gpu": 35.64, + "tokens/trainable": 88297 + }, + { + "epoch": 0.7578125, + "grad_norm": 0.6184028387069702, + "learning_rate": 7.606068977997255e-05, + "loss": 0.026482408866286278, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.02684, + "step": 194, + "tokens/total": 5868704, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 88768 + }, + { + "epoch": 0.76171875, + "grad_norm": 0.17681379616260529, + "learning_rate": 7.580371737159148e-05, + "loss": 0.0036258562467992306, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00363, + "step": 195, + "tokens/total": 5896928, + "tokens/train_per_sec_per_gpu": 30.88, + "tokens/trainable": 89201 + }, + { + "epoch": 0.765625, + "grad_norm": 0.09678306430578232, + "learning_rate": 7.554587923561324e-05, + "loss": 0.0024895307142287493, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00249, + "step": 196, + "tokens/total": 5927328, + "tokens/train_per_sec_per_gpu": 37.84, + "tokens/trainable": 89663 + }, + { + "epoch": 0.76953125, + "grad_norm": 0.033167868852615356, + "learning_rate": 7.528718610173511e-05, + "loss": 0.0012309665326029062, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.34, + "memory/max_allocated (GiB)": 33.34, + "ppl": 1.00123, + "step": 197, + "tokens/total": 5955648, + "tokens/train_per_sec_per_gpu": 38.68, + "tokens/trainable": 90111 + }, + { + "epoch": 0.7734375, + "grad_norm": 0.36096563935279846, + "learning_rate": 7.502764873523431e-05, + "loss": 0.011163209564983845, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01123, + "step": 198, + "tokens/total": 5985904, + "tokens/train_per_sec_per_gpu": 34.73, + "tokens/trainable": 90572 + }, + { + "epoch": 0.77734375, + "grad_norm": 0.07940459251403809, + "learning_rate": 7.476727793652011e-05, + "loss": 0.00193371856585145, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00194, + "step": 199, + "tokens/total": 6016656, + "tokens/train_per_sec_per_gpu": 37.02, + "tokens/trainable": 91051 + }, + { + "epoch": 0.78125, + "grad_norm": 0.28284725546836853, + "learning_rate": 7.450608454068415e-05, + "loss": 0.008390676230192184, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00843, + "step": 200, + "tokens/total": 6046912, + "tokens/train_per_sec_per_gpu": 38.74, + "tokens/trainable": 91527 + }, + { + "epoch": 0.78515625, + "grad_norm": 0.02596334181725979, + "learning_rate": 7.424407941704987e-05, + "loss": 0.0006643222877755761, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00066, + "step": 201, + "tokens/total": 6076912, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 91973 + }, + { + "epoch": 0.7890625, + "grad_norm": 0.26352518796920776, + "learning_rate": 7.398127346871986e-05, + "loss": 0.01052158698439598, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01058, + "step": 202, + "tokens/total": 6107264, + "tokens/train_per_sec_per_gpu": 37.23, + "tokens/trainable": 92438 + }, + { + "epoch": 0.79296875, + "grad_norm": 0.4626573920249939, + "learning_rate": 7.371767763212238e-05, + "loss": 0.026418011635541916, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02677, + "step": 203, + "tokens/total": 6137344, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 92886 + }, + { + "epoch": 0.796875, + "grad_norm": 0.49221205711364746, + "learning_rate": 7.345330287655617e-05, + "loss": 0.005499291233718395, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00551, + "step": 204, + "tokens/total": 6167888, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 93359 + }, + { + "epoch": 0.80078125, + "grad_norm": 0.14302127063274384, + "learning_rate": 7.31881602037339e-05, + "loss": 0.002654140582308173, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00266, + "step": 205, + "tokens/total": 6198080, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 93783 + }, + { + "epoch": 0.8046875, + "grad_norm": 0.1867009699344635, + "learning_rate": 7.29222606473245e-05, + "loss": 0.005218994803726673, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00523, + "step": 206, + "tokens/total": 6228480, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 94262 + }, + { + "epoch": 0.80859375, + "grad_norm": 0.396358460187912, + "learning_rate": 7.265561527249383e-05, + "loss": 0.010595105588436127, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01065, + "step": 207, + "tokens/total": 6258640, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 94753 + }, + { + "epoch": 0.8125, + "grad_norm": 0.04140181094408035, + "learning_rate": 7.238823517544436e-05, + "loss": 0.0008800303330644965, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00088, + "step": 208, + "tokens/total": 6288928, + "tokens/train_per_sec_per_gpu": 34.21, + "tokens/trainable": 95238 + }, + { + "epoch": 0.81640625, + "grad_norm": 0.09654782712459564, + "learning_rate": 7.212013148295333e-05, + "loss": 0.0018917301204055548, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00189, + "step": 209, + "tokens/total": 6319568, + "tokens/train_per_sec_per_gpu": 29.65, + "tokens/trainable": 95639 + }, + { + "epoch": 0.8203125, + "grad_norm": 0.2853289842605591, + "learning_rate": 7.185131535190975e-05, + "loss": 0.005200013052672148, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00521, + "step": 210, + "tokens/total": 6349824, + "tokens/train_per_sec_per_gpu": 35.91, + "tokens/trainable": 96096 + }, + { + "epoch": 0.82421875, + "grad_norm": 1.9312809705734253, + "learning_rate": 7.158179796885005e-05, + "loss": 0.012804300524294376, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01289, + "step": 211, + "tokens/total": 6379920, + "tokens/train_per_sec_per_gpu": 34.64, + "tokens/trainable": 96517 + }, + { + "epoch": 0.828125, + "grad_norm": 0.794370174407959, + "learning_rate": 7.131159054949273e-05, + "loss": 0.024097949266433716, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02439, + "step": 212, + "tokens/total": 6410432, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 96942 + }, + { + "epoch": 0.83203125, + "grad_norm": 0.5890098214149475, + "learning_rate": 7.104070433827139e-05, + "loss": 0.009191717952489853, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00923, + "step": 213, + "tokens/total": 6440864, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 97380 + }, + { + "epoch": 0.8359375, + "grad_norm": 0.17729948461055756, + "learning_rate": 7.076915060786705e-05, + "loss": 0.002657170407474041, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00266, + "step": 214, + "tokens/total": 6471136, + "tokens/train_per_sec_per_gpu": 36.92, + "tokens/trainable": 97869 + }, + { + "epoch": 0.83984375, + "grad_norm": 0.11908628791570663, + "learning_rate": 7.049694065873882e-05, + "loss": 0.0032271994277834892, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00323, + "step": 215, + "tokens/total": 6501328, + "tokens/train_per_sec_per_gpu": 37.9, + "tokens/trainable": 98320 + }, + { + "epoch": 0.84375, + "grad_norm": 0.14903610944747925, + "learning_rate": 7.022408581865382e-05, + "loss": 0.0029319487512111664, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00294, + "step": 216, + "tokens/total": 6531680, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 98793 + }, + { + "epoch": 0.84765625, + "grad_norm": 0.10195992887020111, + "learning_rate": 6.99505974422157e-05, + "loss": 0.0022444254718720913, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00225, + "step": 217, + "tokens/total": 6562032, + "tokens/train_per_sec_per_gpu": 37.33, + "tokens/trainable": 99278 + }, + { + "epoch": 0.8515625, + "grad_norm": 0.07194698601961136, + "learning_rate": 6.967648691039213e-05, + "loss": 0.0016705383313819766, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00167, + "step": 218, + "tokens/total": 6592304, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 99751 + }, + { + "epoch": 0.85546875, + "grad_norm": 0.07926305383443832, + "learning_rate": 6.940176563004123e-05, + "loss": 0.0014279746683314443, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00143, + "step": 219, + "tokens/total": 6622736, + "tokens/train_per_sec_per_gpu": 32.21, + "tokens/trainable": 100184 + }, + { + "epoch": 0.859375, + "grad_norm": 4.5476202964782715, + "learning_rate": 6.912644503343682e-05, + "loss": 0.028542017564177513, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02895, + "step": 220, + "tokens/total": 6653168, + "tokens/train_per_sec_per_gpu": 38.12, + "tokens/trainable": 100656 + }, + { + "epoch": 0.86328125, + "grad_norm": 0.2030973732471466, + "learning_rate": 6.885053657779273e-05, + "loss": 0.002416168339550495, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00242, + "step": 221, + "tokens/total": 6683328, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 101107 + }, + { + "epoch": 0.8671875, + "grad_norm": 0.12359625846147537, + "learning_rate": 6.857405174478604e-05, + "loss": 0.001577290939167142, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00158, + "step": 222, + "tokens/total": 6713840, + "tokens/train_per_sec_per_gpu": 38.57, + "tokens/trainable": 101574 + }, + { + "epoch": 0.87109375, + "grad_norm": 0.6420896053314209, + "learning_rate": 6.82970020400792e-05, + "loss": 0.012318034656345844, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01239, + "step": 223, + "tokens/total": 6744224, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 102015 + }, + { + "epoch": 0.875, + "grad_norm": 0.35973188281059265, + "learning_rate": 6.801939899284132e-05, + "loss": 0.002469731029123068, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00247, + "step": 224, + "tokens/total": 6774432, + "tokens/train_per_sec_per_gpu": 36.04, + "tokens/trainable": 102463 + }, + { + "epoch": 0.87890625, + "grad_norm": 0.10223028063774109, + "learning_rate": 6.774125415526827e-05, + "loss": 0.001325129996985197, + "memory/device_reserved (GiB)": 35.94, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00133, + "step": 225, + "tokens/total": 6804688, + "tokens/train_per_sec_per_gpu": 29.58, + "tokens/trainable": 102880 + }, + { + "epoch": 0.8828125, + "grad_norm": 0.7656016945838928, + "learning_rate": 6.746257910210214e-05, + "loss": 0.01600477285683155, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.01613, + "step": 226, + "tokens/total": 6835376, + "tokens/train_per_sec_per_gpu": 37.94, + "tokens/trainable": 103387 + }, + { + "epoch": 0.88671875, + "grad_norm": 0.2979660928249359, + "learning_rate": 6.718338543014937e-05, + "loss": 0.004661541897803545, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00467, + "step": 227, + "tokens/total": 6865392, + "tokens/train_per_sec_per_gpu": 40.16, + "tokens/trainable": 103880 + }, + { + "epoch": 0.890625, + "grad_norm": 0.4353335499763489, + "learning_rate": 6.69036847577983e-05, + "loss": 0.010676136240363121, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01073, + "step": 228, + "tokens/total": 6895776, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 104343 + }, + { + "epoch": 0.89453125, + "grad_norm": 0.05993308871984482, + "learning_rate": 6.662348872453553e-05, + "loss": 0.0013203206472098827, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.00132, + "step": 229, + "tokens/total": 6925776, + "tokens/train_per_sec_per_gpu": 33.97, + "tokens/trainable": 104808 + }, + { + "epoch": 0.8984375, + "grad_norm": 0.21397961676120758, + "learning_rate": 6.63428089904618e-05, + "loss": 0.00435336958616972, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00436, + "step": 230, + "tokens/total": 6956048, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 105276 + }, + { + "epoch": 0.90234375, + "grad_norm": 0.23647183179855347, + "learning_rate": 6.60616572358065e-05, + "loss": 0.007145303301513195, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00717, + "step": 231, + "tokens/total": 6986384, + "tokens/train_per_sec_per_gpu": 34.34, + "tokens/trainable": 105740 + }, + { + "epoch": 0.90625, + "grad_norm": 0.01726609468460083, + "learning_rate": 6.578004516044172e-05, + "loss": 0.00030698676710017025, + "memory/device_reserved (GiB)": 35.77, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00031, + "step": 232, + "tokens/total": 7016784, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 106216 + }, + { + "epoch": 0.91015625, + "grad_norm": 0.028660962358117104, + "learning_rate": 6.549798448339548e-05, + "loss": 0.000520392379257828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 233, + "tokens/total": 7047184, + "tokens/train_per_sec_per_gpu": 38.58, + "tokens/trainable": 106721 + }, + { + "epoch": 0.9140625, + "grad_norm": 0.1023792177438736, + "learning_rate": 6.521548694236384e-05, + "loss": 0.0018057681154459715, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00181, + "step": 234, + "tokens/total": 7077168, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 107159 + }, + { + "epoch": 0.91796875, + "grad_norm": 0.1957523226737976, + "learning_rate": 6.493256429322259e-05, + "loss": 0.014335989020764828, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01444, + "step": 235, + "tokens/total": 7107616, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 107599 + }, + { + "epoch": 0.921875, + "grad_norm": 0.21969880163669586, + "learning_rate": 6.464922830953799e-05, + "loss": 0.00772607559338212, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00776, + "step": 236, + "tokens/total": 7137728, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 108063 + }, + { + "epoch": 0.92578125, + "grad_norm": 0.04340682178735733, + "learning_rate": 6.436549078207688e-05, + "loss": 0.0009402063442394137, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00094, + "step": 237, + "tokens/total": 7167904, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 108491 + }, + { + "epoch": 0.9296875, + "grad_norm": 0.31009772419929504, + "learning_rate": 6.408136351831592e-05, + "loss": 0.006564700044691563, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00659, + "step": 238, + "tokens/total": 7198288, + "tokens/train_per_sec_per_gpu": 29.57, + "tokens/trainable": 108919 + }, + { + "epoch": 0.93359375, + "grad_norm": 0.6902568340301514, + "learning_rate": 6.379685834195036e-05, + "loss": 0.03366249054670334, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03424, + "step": 239, + "tokens/total": 7228624, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 109360 + }, + { + "epoch": 0.9375, + "grad_norm": 0.07222271710634232, + "learning_rate": 6.351198709240186e-05, + "loss": 0.0014452305622398853, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00145, + "step": 240, + "tokens/total": 7259056, + "tokens/train_per_sec_per_gpu": 32.69, + "tokens/trainable": 109823 + }, + { + "epoch": 0.94140625, + "grad_norm": 0.19805769622325897, + "learning_rate": 6.32267616243259e-05, + "loss": 0.004479828290641308, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00449, + "step": 241, + "tokens/total": 7289328, + "tokens/train_per_sec_per_gpu": 38.97, + "tokens/trainable": 110281 + }, + { + "epoch": 0.9453125, + "grad_norm": 0.15675081312656403, + "learning_rate": 6.294119380711849e-05, + "loss": 0.003017749171704054, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00302, + "step": 242, + "tokens/total": 7319632, + "tokens/train_per_sec_per_gpu": 28.46, + "tokens/trainable": 110694 + }, + { + "epoch": 0.94921875, + "grad_norm": 0.024338258430361748, + "learning_rate": 6.265529552442209e-05, + "loss": 0.000581439642701298, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00058, + "step": 243, + "tokens/total": 7349936, + "tokens/train_per_sec_per_gpu": 39.78, + "tokens/trainable": 111188 + }, + { + "epoch": 0.953125, + "grad_norm": 0.6708511114120483, + "learning_rate": 6.236907867363127e-05, + "loss": 0.011987617239356041, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.01206, + "step": 244, + "tokens/total": 7380576, + "tokens/train_per_sec_per_gpu": 33.59, + "tokens/trainable": 111645 + }, + { + "epoch": 0.95703125, + "grad_norm": 0.05413031578063965, + "learning_rate": 6.208255516539749e-05, + "loss": 0.001512192189693451, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00151, + "step": 245, + "tokens/total": 7411200, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 112128 + }, + { + "epoch": 0.9609375, + "grad_norm": 0.2185056358575821, + "learning_rate": 6.179573692313344e-05, + "loss": 0.002352418377995491, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00236, + "step": 246, + "tokens/total": 7441776, + "tokens/train_per_sec_per_gpu": 33.6, + "tokens/trainable": 112602 + }, + { + "epoch": 0.96484375, + "grad_norm": 0.057806000113487244, + "learning_rate": 6.150863588251694e-05, + "loss": 0.0013613419141620398, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00136, + "step": 247, + "tokens/total": 7470096, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 113056 + }, + { + "epoch": 0.96875, + "grad_norm": 0.39721816778182983, + "learning_rate": 6.122126399099419e-05, + "loss": 0.005575620103627443, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00559, + "step": 248, + "tokens/total": 7500688, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 113497 + }, + { + "epoch": 0.97265625, + "grad_norm": 0.24661272764205933, + "learning_rate": 6.0933633207282615e-05, + "loss": 0.006032303906977177, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00605, + "step": 249, + "tokens/total": 7530832, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 113941 + }, + { + "epoch": 0.9765625, + "grad_norm": 0.0430242083966732, + "learning_rate": 6.064575550087316e-05, + "loss": 0.001341596245765686, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00134, + "step": 250, + "tokens/total": 7559136, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 114373 + }, + { + "epoch": 0.98046875, + "grad_norm": 1.1043962240219116, + "learning_rate": 6.0357642851532245e-05, + "loss": 0.008739812299609184, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00878, + "step": 251, + "tokens/total": 7589648, + "tokens/train_per_sec_per_gpu": 36.97, + "tokens/trainable": 114863 + }, + { + "epoch": 0.984375, + "grad_norm": 0.05025576800107956, + "learning_rate": 6.0069307248803294e-05, + "loss": 0.0008687145891599357, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00087, + "step": 252, + "tokens/total": 7619904, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 115311 + }, + { + "epoch": 0.98828125, + "grad_norm": 0.0689782053232193, + "learning_rate": 5.9780760691507635e-05, + "loss": 0.0014057592488825321, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00141, + "step": 253, + "tokens/total": 7650480, + "tokens/train_per_sec_per_gpu": 32.83, + "tokens/trainable": 115732 + }, + { + "epoch": 0.9921875, + "grad_norm": 0.5817311406135559, + "learning_rate": 5.9492015187245334e-05, + "loss": 0.00938393920660019, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00943, + "step": 254, + "tokens/total": 7680912, + "tokens/train_per_sec_per_gpu": 29.88, + "tokens/trainable": 116164 + }, + { + "epoch": 0.99609375, + "grad_norm": 0.019027426838874817, + "learning_rate": 5.920308275189541e-05, + "loss": 0.000433115113992244, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00043, + "step": 255, + "tokens/total": 7711344, + "tokens/train_per_sec_per_gpu": 37.43, + "tokens/trainable": 116614 + }, + { + "epoch": 1.0, + "grad_norm": 0.05099409446120262, + "learning_rate": 5.8913975409115874e-05, + "loss": 0.0004526513221208006, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00045, + "step": 256, + "tokens/total": 7741936, + "tokens/train_per_sec_per_gpu": 32.59, + "tokens/trainable": 117062 + }, + { + "epoch": 1.00390625, + "grad_norm": 0.09221355617046356, + "learning_rate": 5.8624705189843395e-05, + "loss": 0.0010331927333027124, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00103, + "step": 257, + "tokens/total": 7772272, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 117518 + }, + { + "epoch": 1.0078125, + "grad_norm": 0.037981387227773666, + "learning_rate": 5.833528413179249e-05, + "loss": 0.0006237998022697866, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00062, + "step": 258, + "tokens/total": 7802560, + "tokens/train_per_sec_per_gpu": 30.99, + "tokens/trainable": 117948 + }, + { + "epoch": 1.01171875, + "grad_norm": 0.15115460753440857, + "learning_rate": 5.80457242789548e-05, + "loss": 0.0022220034152269363, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00222, + "step": 259, + "tokens/total": 7832896, + "tokens/train_per_sec_per_gpu": 32.2, + "tokens/trainable": 118388 + }, + { + "epoch": 1.015625, + "grad_norm": 0.03709586337208748, + "learning_rate": 5.77560376810977e-05, + "loss": 0.0004677239339798689, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00047, + "step": 260, + "tokens/total": 7863328, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 118831 + }, + { + "epoch": 1.01953125, + "grad_norm": 0.05125842243432999, + "learning_rate": 5.7466236393263005e-05, + "loss": 0.0011223317123949528, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00112, + "step": 261, + "tokens/total": 7893792, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 119278 + }, + { + "epoch": 1.0234375, + "grad_norm": 0.2065403312444687, + "learning_rate": 5.717633247526522e-05, + "loss": 0.0018837190000340343, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00189, + "step": 262, + "tokens/total": 7924272, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 119787 + }, + { + "epoch": 1.02734375, + "grad_norm": 0.14627555012702942, + "learning_rate": 5.688633799118971e-05, + "loss": 0.0023111533373594284, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00231, + "step": 263, + "tokens/total": 7954624, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 120276 + }, + { + "epoch": 1.03125, + "grad_norm": 0.045769475400447845, + "learning_rate": 5.659626500889066e-05, + "loss": 0.00028873005066998303, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00029, + "step": 264, + "tokens/total": 7985184, + "tokens/train_per_sec_per_gpu": 37.7, + "tokens/trainable": 120753 + }, + { + "epoch": 1.03515625, + "grad_norm": 0.3239908814430237, + "learning_rate": 5.6306125599488905e-05, + "loss": 0.003724604845046997, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00373, + "step": 265, + "tokens/total": 8015520, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 121219 + }, + { + "epoch": 1.0390625, + "grad_norm": 0.6028871536254883, + "learning_rate": 5.601593183686955e-05, + "loss": 0.01656029187142849, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0167, + "step": 266, + "tokens/total": 8046064, + "tokens/train_per_sec_per_gpu": 33.08, + "tokens/trainable": 121694 + }, + { + "epoch": 1.04296875, + "grad_norm": 0.08973632007837296, + "learning_rate": 5.572569579717961e-05, + "loss": 0.0009177342290058732, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00092, + "step": 267, + "tokens/total": 8076352, + "tokens/train_per_sec_per_gpu": 40.77, + "tokens/trainable": 122198 + }, + { + "epoch": 1.046875, + "grad_norm": 0.007786457426846027, + "learning_rate": 5.543542955832538e-05, + "loss": 8.95903940545395e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00009, + "step": 268, + "tokens/total": 8106608, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 122660 + }, + { + "epoch": 1.05078125, + "grad_norm": 0.3484652042388916, + "learning_rate": 5.514514519946986e-05, + "loss": 0.0015085680643096566, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00151, + "step": 269, + "tokens/total": 8136768, + "tokens/train_per_sec_per_gpu": 33.61, + "tokens/trainable": 123120 + }, + { + "epoch": 1.0546875, + "grad_norm": 0.005140791181474924, + "learning_rate": 5.485485480053015e-05, + "loss": 8.239979069912806e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00008, + "step": 270, + "tokens/total": 8167248, + "tokens/train_per_sec_per_gpu": 36.79, + "tokens/trainable": 123610 + }, + { + "epoch": 1.05859375, + "grad_norm": 0.004375405143946409, + "learning_rate": 5.4564570441674645e-05, + "loss": 8.648644870845601e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00009, + "step": 271, + "tokens/total": 8197360, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 124092 + }, + { + "epoch": 1.0625, + "grad_norm": 0.4465174973011017, + "learning_rate": 5.42743042028204e-05, + "loss": 0.004284500610083342, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00429, + "step": 272, + "tokens/total": 8227648, + "tokens/train_per_sec_per_gpu": 35.26, + "tokens/trainable": 124562 + }, + { + "epoch": 1.06640625, + "grad_norm": 0.15294252336025238, + "learning_rate": 5.3984068163130464e-05, + "loss": 0.0017521766712889075, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00175, + "step": 273, + "tokens/total": 8257696, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 124992 + }, + { + "epoch": 1.0703125, + "grad_norm": 0.26586484909057617, + "learning_rate": 5.369387440051111e-05, + "loss": 0.0037963378708809614, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.0038, + "step": 274, + "tokens/total": 8288272, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 125453 + }, + { + "epoch": 1.07421875, + "grad_norm": 0.0579276978969574, + "learning_rate": 5.340373499110935e-05, + "loss": 0.0005296460585668683, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00053, + "step": 275, + "tokens/total": 8318912, + "tokens/train_per_sec_per_gpu": 32.01, + "tokens/trainable": 125904 + }, + { + "epoch": 1.078125, + "grad_norm": 0.12190263718366623, + "learning_rate": 5.3113662008810304e-05, + "loss": 0.0012181613128632307, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00122, + "step": 276, + "tokens/total": 8349280, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 126378 + }, + { + "epoch": 1.08203125, + "grad_norm": 0.06377701461315155, + "learning_rate": 5.282366752473479e-05, + "loss": 0.0005520237027667463, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00055, + "step": 277, + "tokens/total": 8379744, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 126843 + }, + { + "epoch": 1.0859375, + "grad_norm": 0.5330214500427246, + "learning_rate": 5.2533763606737005e-05, + "loss": 0.010966995730996132, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01103, + "step": 278, + "tokens/total": 8409968, + "tokens/train_per_sec_per_gpu": 38.83, + "tokens/trainable": 127346 + }, + { + "epoch": 1.08984375, + "grad_norm": 0.28640905022621155, + "learning_rate": 5.224396231890232e-05, + "loss": 0.017954371869564056, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01812, + "step": 279, + "tokens/total": 8440112, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 127804 + }, + { + "epoch": 1.09375, + "grad_norm": 0.003401878522709012, + "learning_rate": 5.195427572104522e-05, + "loss": 5.5312390031758696e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 280, + "tokens/total": 8470480, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 128253 + }, + { + "epoch": 1.09765625, + "grad_norm": 0.20829863846302032, + "learning_rate": 5.166471586820751e-05, + "loss": 0.0049300119280815125, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00494, + "step": 281, + "tokens/total": 8500752, + "tokens/train_per_sec_per_gpu": 36.8, + "tokens/trainable": 128735 + }, + { + "epoch": 1.1015625, + "grad_norm": 0.010676818899810314, + "learning_rate": 5.1375294810156615e-05, + "loss": 0.00017420266522094607, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00017, + "step": 282, + "tokens/total": 8530944, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 129209 + }, + { + "epoch": 1.10546875, + "grad_norm": 0.1060071736574173, + "learning_rate": 5.1086024590884144e-05, + "loss": 0.0014481758698821068, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00145, + "step": 283, + "tokens/total": 8561312, + "tokens/train_per_sec_per_gpu": 33.5, + "tokens/trainable": 129659 + }, + { + "epoch": 1.109375, + "grad_norm": 0.010597283020615578, + "learning_rate": 5.079691724810461e-05, + "loss": 0.00021654966985806823, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 284, + "tokens/total": 8591712, + "tokens/train_per_sec_per_gpu": 33.14, + "tokens/trainable": 130106 + }, + { + "epoch": 1.11328125, + "grad_norm": 0.3154262900352478, + "learning_rate": 5.0507984812754684e-05, + "loss": 0.00555079523473978, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00557, + "step": 285, + "tokens/total": 8622096, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 130556 + }, + { + "epoch": 1.1171875, + "grad_norm": 0.009663589298725128, + "learning_rate": 5.021923930849237e-05, + "loss": 0.00018602647469379008, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 286, + "tokens/total": 8652304, + "tokens/train_per_sec_per_gpu": 34.84, + "tokens/trainable": 131013 + }, + { + "epoch": 1.12109375, + "grad_norm": 0.02170551009476185, + "learning_rate": 4.99306927511967e-05, + "loss": 0.00036282287328504026, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00036, + "step": 287, + "tokens/total": 8682704, + "tokens/train_per_sec_per_gpu": 34.57, + "tokens/trainable": 131469 + }, + { + "epoch": 1.125, + "grad_norm": 0.1057235449552536, + "learning_rate": 4.964235714846775e-05, + "loss": 0.0015502857277169824, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00155, + "step": 288, + "tokens/total": 8713056, + "tokens/train_per_sec_per_gpu": 36.55, + "tokens/trainable": 131943 + }, + { + "epoch": 1.12890625, + "grad_norm": 0.09496571868658066, + "learning_rate": 4.9354244499126866e-05, + "loss": 0.0017094009090214968, + "memory/device_reserved (GiB)": 35.91, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00171, + "step": 289, + "tokens/total": 8743232, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 132417 + }, + { + "epoch": 1.1328125, + "grad_norm": 0.023917539045214653, + "learning_rate": 4.90663667927174e-05, + "loss": 0.0005100768757984042, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00051, + "step": 290, + "tokens/total": 8773664, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 132866 + }, + { + "epoch": 1.13671875, + "grad_norm": 0.008018841035664082, + "learning_rate": 4.877873600900581e-05, + "loss": 0.00021408281463664025, + "memory/device_reserved (GiB)": 35.56, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00021, + "step": 291, + "tokens/total": 8803920, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 133367 + }, + { + "epoch": 1.140625, + "grad_norm": 0.10013210028409958, + "learning_rate": 4.849136411748306e-05, + "loss": 0.0012666326947510242, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00127, + "step": 292, + "tokens/total": 8834240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 133799 + }, + { + "epoch": 1.14453125, + "grad_norm": 0.04551267251372337, + "learning_rate": 4.8204263076866574e-05, + "loss": 0.0007189091411419213, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00072, + "step": 293, + "tokens/total": 8864560, + "tokens/train_per_sec_per_gpu": 36.86, + "tokens/trainable": 134271 + }, + { + "epoch": 1.1484375, + "grad_norm": 0.028762396425008774, + "learning_rate": 4.791744483460251e-05, + "loss": 0.0005527742905542254, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00055, + "step": 294, + "tokens/total": 8895312, + "tokens/train_per_sec_per_gpu": 34.05, + "tokens/trainable": 134695 + }, + { + "epoch": 1.15234375, + "grad_norm": 0.0814395472407341, + "learning_rate": 4.7630921326368736e-05, + "loss": 0.0011326507665216923, + "memory/device_reserved (GiB)": 35.58, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00113, + "step": 295, + "tokens/total": 8925744, + "tokens/train_per_sec_per_gpu": 36.1, + "tokens/trainable": 135177 + }, + { + "epoch": 1.15625, + "grad_norm": 0.08422354608774185, + "learning_rate": 4.7344704475577916e-05, + "loss": 0.0011638643918558955, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00116, + "step": 296, + "tokens/total": 8956048, + "tokens/train_per_sec_per_gpu": 33.71, + "tokens/trainable": 135611 + }, + { + "epoch": 1.16015625, + "grad_norm": 0.019762732088565826, + "learning_rate": 4.705880619288153e-05, + "loss": 0.00038686307379975915, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00039, + "step": 297, + "tokens/total": 8986320, + "tokens/train_per_sec_per_gpu": 33.09, + "tokens/trainable": 136050 + }, + { + "epoch": 1.1640625, + "grad_norm": 0.019799284636974335, + "learning_rate": 4.677323837567412e-05, + "loss": 0.0003641210787463933, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00036, + "step": 298, + "tokens/total": 9016800, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 136507 + }, + { + "epoch": 1.16796875, + "grad_norm": 0.010453739203512669, + "learning_rate": 4.6488012907598146e-05, + "loss": 0.00018282064411323518, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00018, + "step": 299, + "tokens/total": 9047184, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 136965 + }, + { + "epoch": 1.171875, + "grad_norm": 0.13788308203220367, + "learning_rate": 4.620314165804964e-05, + "loss": 0.002034355653449893, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00204, + "step": 300, + "tokens/total": 9077488, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 137415 + }, + { + "epoch": 1.17578125, + "grad_norm": 0.056629061698913574, + "learning_rate": 4.591863648168407e-05, + "loss": 0.0006158994510769844, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00062, + "step": 301, + "tokens/total": 9107968, + "tokens/train_per_sec_per_gpu": 36.22, + "tokens/trainable": 137910 + }, + { + "epoch": 1.1796875, + "grad_norm": 0.027086833491921425, + "learning_rate": 4.5634509217923135e-05, + "loss": 0.0004739577416330576, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00047, + "step": 302, + "tokens/total": 9138240, + "tokens/train_per_sec_per_gpu": 34.8, + "tokens/trainable": 138386 + }, + { + "epoch": 1.18359375, + "grad_norm": 0.035595983266830444, + "learning_rate": 4.535077169046201e-05, + "loss": 0.00047377185546793044, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00047, + "step": 303, + "tokens/total": 9168640, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 138845 + }, + { + "epoch": 1.1875, + "grad_norm": 0.07773973047733307, + "learning_rate": 4.506743570677743e-05, + "loss": 0.0008522339048795402, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00085, + "step": 304, + "tokens/total": 9198912, + "tokens/train_per_sec_per_gpu": 31.07, + "tokens/trainable": 139270 + }, + { + "epoch": 1.19140625, + "grad_norm": 0.2708052098751068, + "learning_rate": 4.478451305763618e-05, + "loss": 0.017264865338802338, + "memory/device_reserved (GiB)": 35.66, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01741, + "step": 305, + "tokens/total": 9229088, + "tokens/train_per_sec_per_gpu": 33.55, + "tokens/trainable": 139728 + }, + { + "epoch": 1.1953125, + "grad_norm": 0.10748651623725891, + "learning_rate": 4.450201551660454e-05, + "loss": 0.0018072875682264566, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00181, + "step": 306, + "tokens/total": 9259680, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 140219 + }, + { + "epoch": 1.19921875, + "grad_norm": 0.0025890350807458162, + "learning_rate": 4.4219954839558276e-05, + "loss": 4.9693619075696915e-05, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.19, + "memory/max_allocated (GiB)": 33.19, + "ppl": 1.00005, + "step": 307, + "tokens/total": 9287584, + "tokens/train_per_sec_per_gpu": 34.51, + "tokens/trainable": 140657 + }, + { + "epoch": 1.203125, + "grad_norm": 0.018638836219906807, + "learning_rate": 4.393834276419352e-05, + "loss": 0.0003076127031818032, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00031, + "step": 308, + "tokens/total": 9318096, + "tokens/train_per_sec_per_gpu": 30.38, + "tokens/trainable": 141108 + }, + { + "epoch": 1.20703125, + "grad_norm": 0.010553287342190742, + "learning_rate": 4.36571910095382e-05, + "loss": 0.00020812047296203673, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00021, + "step": 309, + "tokens/total": 9348544, + "tokens/train_per_sec_per_gpu": 29.44, + "tokens/trainable": 141531 + }, + { + "epoch": 1.2109375, + "grad_norm": 0.2629532217979431, + "learning_rate": 4.337651127546448e-05, + "loss": 0.023792965337634087, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02408, + "step": 310, + "tokens/total": 9378848, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 141965 + }, + { + "epoch": 1.21484375, + "grad_norm": 0.01139355544000864, + "learning_rate": 4.3096315242201736e-05, + "loss": 0.00018112164980266243, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00018, + "step": 311, + "tokens/total": 9409104, + "tokens/train_per_sec_per_gpu": 34.52, + "tokens/trainable": 142429 + }, + { + "epoch": 1.21875, + "grad_norm": 0.01917850784957409, + "learning_rate": 4.2816614569850635e-05, + "loss": 0.00016473264258820564, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00016, + "step": 312, + "tokens/total": 9439264, + "tokens/train_per_sec_per_gpu": 38.76, + "tokens/trainable": 142895 + }, + { + "epoch": 1.22265625, + "grad_norm": 0.4227152466773987, + "learning_rate": 4.2537420897897864e-05, + "loss": 0.004962727427482605, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00498, + "step": 313, + "tokens/total": 9469648, + "tokens/train_per_sec_per_gpu": 37.12, + "tokens/trainable": 143370 + }, + { + "epoch": 1.2265625, + "grad_norm": 0.09607069194316864, + "learning_rate": 4.225874584473174e-05, + "loss": 0.0009555588476359844, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00096, + "step": 314, + "tokens/total": 9499968, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 143826 + }, + { + "epoch": 1.23046875, + "grad_norm": 0.0108202388510108, + "learning_rate": 4.19806010071587e-05, + "loss": 0.0002800720394589007, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00028, + "step": 315, + "tokens/total": 9530448, + "tokens/train_per_sec_per_gpu": 35.11, + "tokens/trainable": 144308 + }, + { + "epoch": 1.234375, + "grad_norm": 0.5568331480026245, + "learning_rate": 4.170299795992081e-05, + "loss": 0.0007924925885163248, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00079, + "step": 316, + "tokens/total": 9560848, + "tokens/train_per_sec_per_gpu": 36.0, + "tokens/trainable": 144765 + }, + { + "epoch": 1.23828125, + "grad_norm": 0.3749312162399292, + "learning_rate": 4.142594825521398e-05, + "loss": 0.009996296837925911, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01005, + "step": 317, + "tokens/total": 9591376, + "tokens/train_per_sec_per_gpu": 37.75, + "tokens/trainable": 145265 + }, + { + "epoch": 1.2421875, + "grad_norm": 0.22495749592781067, + "learning_rate": 4.114946342220728e-05, + "loss": 0.002975808223709464, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00298, + "step": 318, + "tokens/total": 9621728, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 145724 + }, + { + "epoch": 1.24609375, + "grad_norm": 0.47026923298835754, + "learning_rate": 4.087355496656321e-05, + "loss": 0.006408916786313057, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00643, + "step": 319, + "tokens/total": 9652272, + "tokens/train_per_sec_per_gpu": 37.72, + "tokens/trainable": 146219 + }, + { + "epoch": 1.25, + "grad_norm": 0.17889857292175293, + "learning_rate": 4.05982343699588e-05, + "loss": 0.0023502246476709843, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00235, + "step": 320, + "tokens/total": 9682672, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 146693 + }, + { + "epoch": 1.25390625, + "grad_norm": 0.062361083924770355, + "learning_rate": 4.0323513089607876e-05, + "loss": 0.0006638169870711863, + "memory/device_reserved (GiB)": 35.9, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00066, + "step": 321, + "tokens/total": 9712832, + "tokens/train_per_sec_per_gpu": 35.87, + "tokens/trainable": 147139 + }, + { + "epoch": 1.2578125, + "grad_norm": 0.009022507816553116, + "learning_rate": 4.004940255778431e-05, + "loss": 0.00017464160919189453, + "memory/device_reserved (GiB)": 36.14, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00017, + "step": 322, + "tokens/total": 9743344, + "tokens/train_per_sec_per_gpu": 35.74, + "tokens/trainable": 147589 + }, + { + "epoch": 1.26171875, + "grad_norm": 0.38307738304138184, + "learning_rate": 3.977591418134619e-05, + "loss": 0.0034184439573436975, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00342, + "step": 323, + "tokens/total": 9773760, + "tokens/train_per_sec_per_gpu": 35.77, + "tokens/trainable": 148045 + }, + { + "epoch": 1.265625, + "grad_norm": 0.3203769028186798, + "learning_rate": 3.95030593412612e-05, + "loss": 0.0051149846985936165, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00513, + "step": 324, + "tokens/total": 9804080, + "tokens/train_per_sec_per_gpu": 37.47, + "tokens/trainable": 148494 + }, + { + "epoch": 1.26953125, + "grad_norm": 0.7488558292388916, + "learning_rate": 3.923084939213296e-05, + "loss": 0.01595677062869072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.01608, + "step": 325, + "tokens/total": 9834400, + "tokens/train_per_sec_per_gpu": 36.84, + "tokens/trainable": 148957 + }, + { + "epoch": 1.2734375, + "grad_norm": 0.009091824293136597, + "learning_rate": 3.895929566172861e-05, + "loss": 0.00014956737868487835, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00015, + "step": 326, + "tokens/total": 9864704, + "tokens/train_per_sec_per_gpu": 38.41, + "tokens/trainable": 149451 + }, + { + "epoch": 1.27734375, + "grad_norm": 0.005110942758619785, + "learning_rate": 3.868840945050728e-05, + "loss": 0.00013913170550949872, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00014, + "step": 327, + "tokens/total": 9894656, + "tokens/train_per_sec_per_gpu": 40.07, + "tokens/trainable": 149904 + }, + { + "epoch": 1.28125, + "grad_norm": 0.007466079201549292, + "learning_rate": 3.841820203114995e-05, + "loss": 0.00013769841461908072, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00014, + "step": 328, + "tokens/total": 9924976, + "tokens/train_per_sec_per_gpu": 34.36, + "tokens/trainable": 150378 + }, + { + "epoch": 1.28515625, + "grad_norm": 0.16560006141662598, + "learning_rate": 3.814868464809027e-05, + "loss": 0.009101686999201775, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00914, + "step": 329, + "tokens/total": 9955376, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 150853 + }, + { + "epoch": 1.2890625, + "grad_norm": 0.01060777809470892, + "learning_rate": 3.787986851704667e-05, + "loss": 0.00019600678933784366, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0002, + "step": 330, + "tokens/total": 9985696, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 151289 + }, + { + "epoch": 1.29296875, + "grad_norm": 0.0109481830149889, + "learning_rate": 3.7611764824555654e-05, + "loss": 0.00029803148936480284, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.0003, + "step": 331, + "tokens/total": 10016112, + "tokens/train_per_sec_per_gpu": 37.68, + "tokens/trainable": 151757 + }, + { + "epoch": 1.296875, + "grad_norm": 0.011753400787711143, + "learning_rate": 3.734438472750619e-05, + "loss": 0.00021420676785055548, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00021, + "step": 332, + "tokens/total": 10046544, + "tokens/train_per_sec_per_gpu": 34.55, + "tokens/trainable": 152207 + }, + { + "epoch": 1.30078125, + "grad_norm": 0.3032369613647461, + "learning_rate": 3.707773935267552e-05, + "loss": 0.004725611303001642, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00474, + "step": 333, + "tokens/total": 10077120, + "tokens/train_per_sec_per_gpu": 32.34, + "tokens/trainable": 152646 + }, + { + "epoch": 1.3046875, + "grad_norm": 0.024216441437602043, + "learning_rate": 3.68118397962661e-05, + "loss": 0.0005056065274402499, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00051, + "step": 334, + "tokens/total": 10107600, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 153063 + }, + { + "epoch": 1.30859375, + "grad_norm": 0.10152052342891693, + "learning_rate": 3.654669712344384e-05, + "loss": 0.001724423374980688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00173, + "step": 335, + "tokens/total": 10137712, + "tokens/train_per_sec_per_gpu": 30.92, + "tokens/trainable": 153493 + }, + { + "epoch": 1.3125, + "grad_norm": 0.06117634102702141, + "learning_rate": 3.628232236787763e-05, + "loss": 0.0013021706836298108, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0013, + "step": 336, + "tokens/total": 10168096, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 153945 + }, + { + "epoch": 1.31640625, + "grad_norm": 0.022510197013616562, + "learning_rate": 3.6018726531280144e-05, + "loss": 0.0003054860280826688, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00031, + "step": 337, + "tokens/total": 10198720, + "tokens/train_per_sec_per_gpu": 35.4, + "tokens/trainable": 154432 + }, + { + "epoch": 1.3203125, + "grad_norm": 0.19209441542625427, + "learning_rate": 3.575592058295017e-05, + "loss": 0.0020988616161048412, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0021, + "step": 338, + "tokens/total": 10228976, + "tokens/train_per_sec_per_gpu": 34.47, + "tokens/trainable": 154881 + }, + { + "epoch": 1.32421875, + "grad_norm": 0.008805932477116585, + "learning_rate": 3.549391545931585e-05, + "loss": 0.00020574698282871395, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00021, + "step": 339, + "tokens/total": 10259440, + "tokens/train_per_sec_per_gpu": 30.32, + "tokens/trainable": 155306 + }, + { + "epoch": 1.328125, + "grad_norm": 0.01874561607837677, + "learning_rate": 3.5232722063479914e-05, + "loss": 0.0003597445320338011, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00036, + "step": 340, + "tokens/total": 10289824, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 155719 + }, + { + "epoch": 1.33203125, + "grad_norm": 0.005896273069083691, + "learning_rate": 3.49723512647657e-05, + "loss": 0.00010345762711949646, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 341, + "tokens/total": 10320144, + "tokens/train_per_sec_per_gpu": 36.3, + "tokens/trainable": 156209 + }, + { + "epoch": 1.3359375, + "grad_norm": 0.010024541057646275, + "learning_rate": 3.471281389826491e-05, + "loss": 0.0001570192980580032, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00016, + "step": 342, + "tokens/total": 10350528, + "tokens/train_per_sec_per_gpu": 34.17, + "tokens/trainable": 156656 + }, + { + "epoch": 1.33984375, + "grad_norm": 0.10390307009220123, + "learning_rate": 3.4454120764386764e-05, + "loss": 0.0013503385707736015, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00135, + "step": 343, + "tokens/total": 10380816, + "tokens/train_per_sec_per_gpu": 34.23, + "tokens/trainable": 157118 + }, + { + "epoch": 1.34375, + "grad_norm": 0.049824248999357224, + "learning_rate": 3.4196282628408526e-05, + "loss": 0.0005016025970689952, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0005, + "step": 344, + "tokens/total": 10411312, + "tokens/train_per_sec_per_gpu": 33.56, + "tokens/trainable": 157562 + }, + { + "epoch": 1.34765625, + "grad_norm": 0.011971387080848217, + "learning_rate": 3.3939310220027456e-05, + "loss": 0.00018647368415258825, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00019, + "step": 345, + "tokens/total": 10441696, + "tokens/train_per_sec_per_gpu": 35.7, + "tokens/trainable": 158003 + }, + { + "epoch": 1.3515625, + "grad_norm": 0.020012276247143745, + "learning_rate": 3.3683214232914404e-05, + "loss": 0.0004137884243391454, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00041, + "step": 346, + "tokens/total": 10472032, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 158419 + }, + { + "epoch": 1.35546875, + "grad_norm": 0.1378275603055954, + "learning_rate": 3.342800532426873e-05, + "loss": 0.001066669705323875, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00107, + "step": 347, + "tokens/total": 10502176, + "tokens/train_per_sec_per_gpu": 36.17, + "tokens/trainable": 158877 + }, + { + "epoch": 1.359375, + "grad_norm": 0.012419048696756363, + "learning_rate": 3.317369411437484e-05, + "loss": 0.0002563716552685946, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.0, + "memory/max_allocated (GiB)": 34.0, + "ppl": 1.00026, + "step": 348, + "tokens/total": 10532896, + "tokens/train_per_sec_per_gpu": 29.19, + "tokens/trainable": 159299 + }, + { + "epoch": 1.36328125, + "grad_norm": 0.0319787822663784, + "learning_rate": 3.292029118616024e-05, + "loss": 0.0005879281088709831, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00059, + "step": 349, + "tokens/total": 10563024, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 159750 + }, + { + "epoch": 1.3671875, + "grad_norm": 0.009861056692898273, + "learning_rate": 3.266780708475511e-05, + "loss": 0.00010148907313123345, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.0001, + "step": 350, + "tokens/total": 10593536, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 160217 + }, + { + "epoch": 1.37109375, + "grad_norm": 0.005884335841983557, + "learning_rate": 3.241625231705354e-05, + "loss": 0.00010095423203893006, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.0001, + "step": 351, + "tokens/total": 10623776, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 160675 + }, + { + "epoch": 1.375, + "grad_norm": 0.24247600138187408, + "learning_rate": 3.216563735127618e-05, + "loss": 0.0036722179502248764, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00368, + "step": 352, + "tokens/total": 10654032, + "tokens/train_per_sec_per_gpu": 35.02, + "tokens/trainable": 161132 + }, + { + "epoch": 1.37890625, + "grad_norm": 0.0033956863917410374, + "learning_rate": 3.191597261653475e-05, + "loss": 6.495713023468852e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00006, + "step": 353, + "tokens/total": 10684352, + "tokens/train_per_sec_per_gpu": 29.4, + "tokens/trainable": 161585 + }, + { + "epoch": 1.3828125, + "grad_norm": 0.8877756595611572, + "learning_rate": 3.166726850239794e-05, + "loss": 0.008384269662201405, + "memory/device_reserved (GiB)": 35.41, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00842, + "step": 354, + "tokens/total": 10714816, + "tokens/train_per_sec_per_gpu": 37.32, + "tokens/trainable": 162050 + }, + { + "epoch": 1.38671875, + "grad_norm": 0.00449851481243968, + "learning_rate": 3.141953535845912e-05, + "loss": 8.070325566222891e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00008, + "step": 355, + "tokens/total": 10745296, + "tokens/train_per_sec_per_gpu": 32.65, + "tokens/trainable": 162478 + }, + { + "epoch": 1.390625, + "grad_norm": 0.0048718261532485485, + "learning_rate": 3.11727834939056e-05, + "loss": 7.578312943223864e-05, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.61, + "memory/max_allocated (GiB)": 33.61, + "ppl": 1.00008, + "step": 356, + "tokens/total": 10775296, + "tokens/train_per_sec_per_gpu": 28.74, + "tokens/trainable": 162893 + }, + { + "epoch": 1.39453125, + "grad_norm": 0.21484895050525665, + "learning_rate": 3.092702317708967e-05, + "loss": 0.0030353569891303778, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00304, + "step": 357, + "tokens/total": 10805600, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 163347 + }, + { + "epoch": 1.3984375, + "grad_norm": 0.07261230796575546, + "learning_rate": 3.0682264635101276e-05, + "loss": 0.0004826942749787122, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00048, + "step": 358, + "tokens/total": 10835760, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 163796 + }, + { + "epoch": 1.40234375, + "grad_norm": 0.022041240707039833, + "learning_rate": 3.0438518053342407e-05, + "loss": 0.00020454936020541936, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.0002, + "step": 359, + "tokens/total": 10866208, + "tokens/train_per_sec_per_gpu": 32.47, + "tokens/trainable": 164237 + }, + { + "epoch": 1.40625, + "grad_norm": 0.005708560813218355, + "learning_rate": 3.0195793575103266e-05, + "loss": 0.00012487702770158648, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00012, + "step": 360, + "tokens/total": 10896672, + "tokens/train_per_sec_per_gpu": 34.69, + "tokens/trainable": 164684 + }, + { + "epoch": 1.41015625, + "grad_norm": 0.02250049076974392, + "learning_rate": 2.9954101301140146e-05, + "loss": 0.00025320789427496493, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00025, + "step": 361, + "tokens/total": 10926944, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 165146 + }, + { + "epoch": 1.4140625, + "grad_norm": 0.013911883346736431, + "learning_rate": 2.9713451289255123e-05, + "loss": 0.00022382009774446487, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00022, + "step": 362, + "tokens/total": 10957456, + "tokens/train_per_sec_per_gpu": 35.37, + "tokens/trainable": 165632 + }, + { + "epoch": 1.41796875, + "grad_norm": 0.2468303143978119, + "learning_rate": 2.9473853553877484e-05, + "loss": 0.01283702440559864, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.01292, + "step": 363, + "tokens/total": 10987696, + "tokens/train_per_sec_per_gpu": 34.66, + "tokens/trainable": 166080 + }, + { + "epoch": 1.421875, + "grad_norm": 0.008053003810346127, + "learning_rate": 2.9235318065647e-05, + "loss": 0.00013117909838911146, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00013, + "step": 364, + "tokens/total": 11017904, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 166531 + }, + { + "epoch": 1.42578125, + "grad_norm": 0.04743769019842148, + "learning_rate": 2.8997854750998964e-05, + "loss": 0.00024472299264743924, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00024, + "step": 365, + "tokens/total": 11048224, + "tokens/train_per_sec_per_gpu": 36.91, + "tokens/trainable": 167019 + }, + { + "epoch": 1.4296875, + "grad_norm": 0.008861004374921322, + "learning_rate": 2.8761473491751258e-05, + "loss": 0.00010162356193177402, + "memory/device_reserved (GiB)": 35.84, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0001, + "step": 366, + "tokens/total": 11078544, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 167469 + }, + { + "epoch": 1.43359375, + "grad_norm": 0.06953191757202148, + "learning_rate": 2.8526184124692883e-05, + "loss": 0.0006842018919996917, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00068, + "step": 367, + "tokens/total": 11109088, + "tokens/train_per_sec_per_gpu": 31.86, + "tokens/trainable": 167914 + }, + { + "epoch": 1.4375, + "grad_norm": 0.10965435951948166, + "learning_rate": 2.829199644117484e-05, + "loss": 0.0006139783654361963, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00061, + "step": 368, + "tokens/total": 11139408, + "tokens/train_per_sec_per_gpu": 36.16, + "tokens/trainable": 168357 + }, + { + "epoch": 1.44140625, + "grad_norm": 0.013294316828250885, + "learning_rate": 2.8058920186702553e-05, + "loss": 0.0001493502495577559, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00015, + "step": 369, + "tokens/total": 11169920, + "tokens/train_per_sec_per_gpu": 32.75, + "tokens/trainable": 168832 + }, + { + "epoch": 1.4453125, + "grad_norm": 0.7463682889938354, + "learning_rate": 2.782696506053033e-05, + "loss": 0.013401923701167107, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.01349, + "step": 370, + "tokens/total": 11200288, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 169333 + }, + { + "epoch": 1.44921875, + "grad_norm": 0.001093173515982926, + "learning_rate": 2.7596140715257824e-05, + "loss": 2.9635479222633876e-05, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00003, + "step": 371, + "tokens/total": 11230832, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 169774 + }, + { + "epoch": 1.453125, + "grad_norm": 0.1324324756860733, + "learning_rate": 2.7366456756428184e-05, + "loss": 0.0015135211870074272, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00151, + "step": 372, + "tokens/total": 11261264, + "tokens/train_per_sec_per_gpu": 33.41, + "tokens/trainable": 170241 + }, + { + "epoch": 1.45703125, + "grad_norm": 0.006301951594650745, + "learning_rate": 2.7137922742128486e-05, + "loss": 0.0001302927266806364, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00013, + "step": 373, + "tokens/total": 11291456, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 170706 + }, + { + "epoch": 1.4609375, + "grad_norm": 0.011703136377036572, + "learning_rate": 2.691054818259188e-05, + "loss": 0.00020681106252595782, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00021, + "step": 374, + "tokens/total": 11321952, + "tokens/train_per_sec_per_gpu": 32.53, + "tokens/trainable": 171182 + }, + { + "epoch": 1.46484375, + "grad_norm": 0.015446359291672707, + "learning_rate": 2.6684342539801933e-05, + "loss": 0.0003060699673369527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00031, + "step": 375, + "tokens/total": 11352096, + "tokens/train_per_sec_per_gpu": 34.75, + "tokens/trainable": 171646 + }, + { + "epoch": 1.46875, + "grad_norm": 0.008862881921231747, + "learning_rate": 2.645931522709877e-05, + "loss": 0.00019395005074329674, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00019, + "step": 376, + "tokens/total": 11382528, + "tokens/train_per_sec_per_gpu": 28.57, + "tokens/trainable": 172063 + }, + { + "epoch": 1.47265625, + "grad_norm": 0.03388524800539017, + "learning_rate": 2.6235475608787365e-05, + "loss": 0.0004883318324573338, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00049, + "step": 377, + "tokens/total": 11412832, + "tokens/train_per_sec_per_gpu": 36.63, + "tokens/trainable": 172539 + }, + { + "epoch": 1.4765625, + "grad_norm": 0.02546733431518078, + "learning_rate": 2.6012832999747916e-05, + "loss": 0.0005159692373126745, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00052, + "step": 378, + "tokens/total": 11442864, + "tokens/train_per_sec_per_gpu": 34.99, + "tokens/trainable": 172997 + }, + { + "epoch": 1.48046875, + "grad_norm": 0.045234713703393936, + "learning_rate": 2.579139666504821e-05, + "loss": 0.0004851966805290431, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 379, + "tokens/total": 11473328, + "tokens/train_per_sec_per_gpu": 36.26, + "tokens/trainable": 173481 + }, + { + "epoch": 1.484375, + "grad_norm": 0.011559339240193367, + "learning_rate": 2.557117581955798e-05, + "loss": 0.00026303555932827294, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00026, + "step": 380, + "tokens/total": 11503504, + "tokens/train_per_sec_per_gpu": 30.94, + "tokens/trainable": 173946 + }, + { + "epoch": 1.48828125, + "grad_norm": 0.01947280764579773, + "learning_rate": 2.5352179627565532e-05, + "loss": 0.0004145601997151971, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00041, + "step": 381, + "tokens/total": 11533920, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 174417 + }, + { + "epoch": 1.4921875, + "grad_norm": 0.03896784782409668, + "learning_rate": 2.5134417202396277e-05, + "loss": 0.0005209866212680936, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00052, + "step": 382, + "tokens/total": 11564128, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 174872 + }, + { + "epoch": 1.49609375, + "grad_norm": 0.012542990036308765, + "learning_rate": 2.491789760603361e-05, + "loss": 0.00026536238146945834, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.00027, + "step": 383, + "tokens/total": 11594672, + "tokens/train_per_sec_per_gpu": 31.99, + "tokens/trainable": 175313 + }, + { + "epoch": 1.5, + "grad_norm": 0.23138198256492615, + "learning_rate": 2.4702629848741764e-05, + "loss": 0.011072758585214615, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.01113, + "step": 384, + "tokens/total": 11625056, + "tokens/train_per_sec_per_gpu": 32.96, + "tokens/trainable": 175750 + }, + { + "epoch": 1.50390625, + "grad_norm": 0.5275915861129761, + "learning_rate": 2.4488622888690785e-05, + "loss": 0.016837235540151596, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.01698, + "step": 385, + "tokens/total": 11655488, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 176232 + }, + { + "epoch": 1.5078125, + "grad_norm": 0.008643269538879395, + "learning_rate": 2.427588563158384e-05, + "loss": 0.00011418825306463987, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 386, + "tokens/total": 11685968, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 176680 + }, + { + "epoch": 1.51171875, + "grad_norm": 0.00684578949585557, + "learning_rate": 2.406442693028651e-05, + "loss": 0.0001287878112634644, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 387, + "tokens/total": 11716176, + "tokens/train_per_sec_per_gpu": 33.24, + "tokens/trainable": 177144 + }, + { + "epoch": 1.515625, + "grad_norm": 0.07777013629674911, + "learning_rate": 2.3854255584458547e-05, + "loss": 0.001054117688909173, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00105, + "step": 388, + "tokens/total": 11746704, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 177630 + }, + { + "epoch": 1.51953125, + "grad_norm": 0.021813517436385155, + "learning_rate": 2.3645380340187508e-05, + "loss": 0.00045925029553472996, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00046, + "step": 389, + "tokens/total": 11777008, + "tokens/train_per_sec_per_gpu": 33.51, + "tokens/trainable": 178079 + }, + { + "epoch": 1.5234375, + "grad_norm": 0.2766229808330536, + "learning_rate": 2.3437809889624914e-05, + "loss": 0.0063839266076684, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0064, + "step": 390, + "tokens/total": 11807168, + "tokens/train_per_sec_per_gpu": 36.99, + "tokens/trainable": 178552 + }, + { + "epoch": 1.52734375, + "grad_norm": 0.5076630711555481, + "learning_rate": 2.3231552870624487e-05, + "loss": 0.003972330130636692, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00398, + "step": 391, + "tokens/total": 11837776, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 179021 + }, + { + "epoch": 1.53125, + "grad_norm": 0.046427834779024124, + "learning_rate": 2.3026617866382657e-05, + "loss": 0.00048776037874631584, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00049, + "step": 392, + "tokens/total": 11867968, + "tokens/train_per_sec_per_gpu": 33.57, + "tokens/trainable": 179478 + }, + { + "epoch": 1.53515625, + "grad_norm": 0.12148728966712952, + "learning_rate": 2.2823013405081507e-05, + "loss": 0.00316976523026824, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00317, + "step": 393, + "tokens/total": 11898320, + "tokens/train_per_sec_per_gpu": 33.27, + "tokens/trainable": 179910 + }, + { + "epoch": 1.5390625, + "grad_norm": 0.02432228811085224, + "learning_rate": 2.2620747959533722e-05, + "loss": 0.00029762828489765525, + "memory/device_reserved (GiB)": 35.87, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.0003, + "step": 394, + "tokens/total": 11928624, + "tokens/train_per_sec_per_gpu": 34.45, + "tokens/trainable": 180361 + }, + { + "epoch": 1.54296875, + "grad_norm": 0.01567767933011055, + "learning_rate": 2.2419829946830123e-05, + "loss": 0.00036671021371148527, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00037, + "step": 395, + "tokens/total": 11959088, + "tokens/train_per_sec_per_gpu": 32.31, + "tokens/trainable": 180774 + }, + { + "epoch": 1.546875, + "grad_norm": 0.02068573608994484, + "learning_rate": 2.2220267727989325e-05, + "loss": 0.00030358770163729787, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0003, + "step": 396, + "tokens/total": 11989504, + "tokens/train_per_sec_per_gpu": 35.69, + "tokens/trainable": 181246 + }, + { + "epoch": 1.55078125, + "grad_norm": 0.2614375948905945, + "learning_rate": 2.202206960760984e-05, + "loss": 0.00875169225037098, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00879, + "step": 397, + "tokens/total": 12019904, + "tokens/train_per_sec_per_gpu": 33.65, + "tokens/trainable": 181734 + }, + { + "epoch": 1.5546875, + "grad_norm": 0.037434663623571396, + "learning_rate": 2.182524383352446e-05, + "loss": 0.00044167632586322725, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00044, + "step": 398, + "tokens/total": 12050384, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 182169 + }, + { + "epoch": 1.55859375, + "grad_norm": 0.16984692215919495, + "learning_rate": 2.1629798596457056e-05, + "loss": 0.0007633643108420074, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00076, + "step": 399, + "tokens/total": 12080576, + "tokens/train_per_sec_per_gpu": 35.34, + "tokens/trainable": 182641 + }, + { + "epoch": 1.5625, + "grad_norm": 0.17288610339164734, + "learning_rate": 2.1435742029681725e-05, + "loss": 0.0036359927617013454, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.00364, + "step": 400, + "tokens/total": 12110896, + "tokens/train_per_sec_per_gpu": 26.93, + "tokens/trainable": 183056 + }, + { + "epoch": 1.56640625, + "grad_norm": 0.056392524391412735, + "learning_rate": 2.124308220868431e-05, + "loss": 0.0008229271625168622, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.71, + "memory/max_allocated (GiB)": 33.71, + "ppl": 1.00082, + "step": 401, + "tokens/total": 12140784, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 183504 + }, + { + "epoch": 1.5703125, + "grad_norm": 0.0365372970700264, + "learning_rate": 2.105182715082638e-05, + "loss": 0.0007215660298243165, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00072, + "step": 402, + "tokens/total": 12171024, + "tokens/train_per_sec_per_gpu": 39.29, + "tokens/trainable": 184004 + }, + { + "epoch": 1.57421875, + "grad_norm": 0.016568642109632492, + "learning_rate": 2.0861984815011552e-05, + "loss": 0.00029834112501703203, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.0003, + "step": 403, + "tokens/total": 12201184, + "tokens/train_per_sec_per_gpu": 33.04, + "tokens/trainable": 184476 + }, + { + "epoch": 1.578125, + "grad_norm": 0.06592518836259842, + "learning_rate": 2.0673563101354323e-05, + "loss": 0.0007513194577768445, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 404, + "tokens/total": 12231232, + "tokens/train_per_sec_per_gpu": 32.35, + "tokens/trainable": 184905 + }, + { + "epoch": 1.58203125, + "grad_norm": 0.013345538638532162, + "learning_rate": 2.0486569850851317e-05, + "loss": 0.0001536268973723054, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.00015, + "step": 405, + "tokens/total": 12261280, + "tokens/train_per_sec_per_gpu": 29.82, + "tokens/trainable": 185338 + }, + { + "epoch": 1.5859375, + "grad_norm": 0.11538074165582657, + "learning_rate": 2.0301012845054956e-05, + "loss": 0.001165087684057653, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00117, + "step": 406, + "tokens/total": 12291520, + "tokens/train_per_sec_per_gpu": 32.02, + "tokens/trainable": 185792 + }, + { + "epoch": 1.58984375, + "grad_norm": 0.01569611392915249, + "learning_rate": 2.011689980574966e-05, + "loss": 0.00031081197084859014, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00031, + "step": 407, + "tokens/total": 12321824, + "tokens/train_per_sec_per_gpu": 37.41, + "tokens/trainable": 186251 + }, + { + "epoch": 1.59375, + "grad_norm": 0.02253701724112034, + "learning_rate": 1.993423839463052e-05, + "loss": 0.00034153330489061773, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00034, + "step": 408, + "tokens/total": 12352176, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 186689 + }, + { + "epoch": 1.59765625, + "grad_norm": 0.40578708052635193, + "learning_rate": 1.975303621298445e-05, + "loss": 0.005957326851785183, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00598, + "step": 409, + "tokens/total": 12382608, + "tokens/train_per_sec_per_gpu": 34.7, + "tokens/trainable": 187121 + }, + { + "epoch": 1.6015625, + "grad_norm": 0.15000925958156586, + "learning_rate": 1.957330080137385e-05, + "loss": 0.0013339307624846697, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00133, + "step": 410, + "tokens/total": 12413152, + "tokens/train_per_sec_per_gpu": 35.55, + "tokens/trainable": 187568 + }, + { + "epoch": 1.60546875, + "grad_norm": 0.006941859144717455, + "learning_rate": 1.9395039639322864e-05, + "loss": 0.00013002814375795424, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00013, + "step": 411, + "tokens/total": 12443600, + "tokens/train_per_sec_per_gpu": 34.76, + "tokens/trainable": 188010 + }, + { + "epoch": 1.609375, + "grad_norm": 0.12384258210659027, + "learning_rate": 1.9218260145006073e-05, + "loss": 0.0010012347484007478, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.001, + "step": 412, + "tokens/total": 12471952, + "tokens/train_per_sec_per_gpu": 39.3, + "tokens/trainable": 188446 + }, + { + "epoch": 1.61328125, + "grad_norm": 0.3418160378932953, + "learning_rate": 1.904296967493982e-05, + "loss": 0.00740136718377471, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00743, + "step": 413, + "tokens/total": 12502480, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 188897 + }, + { + "epoch": 1.6171875, + "grad_norm": 0.006758023519068956, + "learning_rate": 1.8869175523676064e-05, + "loss": 6.429506174754351e-05, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00006, + "step": 414, + "tokens/total": 12532720, + "tokens/train_per_sec_per_gpu": 37.49, + "tokens/trainable": 189366 + }, + { + "epoch": 1.62109375, + "grad_norm": 0.19964486360549927, + "learning_rate": 1.869688492349885e-05, + "loss": 0.005143567454069853, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00516, + "step": 415, + "tokens/total": 12562848, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 189779 + }, + { + "epoch": 1.625, + "grad_norm": 0.16783277690410614, + "learning_rate": 1.85261050441233e-05, + "loss": 0.0007222781423479319, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00072, + "step": 416, + "tokens/total": 12593328, + "tokens/train_per_sec_per_gpu": 34.56, + "tokens/trainable": 190245 + }, + { + "epoch": 1.62890625, + "grad_norm": 0.018791699782013893, + "learning_rate": 1.8356842992397304e-05, + "loss": 0.0002913455246016383, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00029, + "step": 417, + "tokens/total": 12623760, + "tokens/train_per_sec_per_gpu": 35.59, + "tokens/trainable": 190737 + }, + { + "epoch": 1.6328125, + "grad_norm": 0.09821721911430359, + "learning_rate": 1.8189105812005714e-05, + "loss": 0.0005217839498072863, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00052, + "step": 418, + "tokens/total": 12654272, + "tokens/train_per_sec_per_gpu": 31.38, + "tokens/trainable": 191155 + }, + { + "epoch": 1.63671875, + "grad_norm": 0.071963369846344, + "learning_rate": 1.802290048317732e-05, + "loss": 0.0007684896700084209, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00077, + "step": 419, + "tokens/total": 12684544, + "tokens/train_per_sec_per_gpu": 31.98, + "tokens/trainable": 191577 + }, + { + "epoch": 1.640625, + "grad_norm": 0.041365377604961395, + "learning_rate": 1.785823392239424e-05, + "loss": 0.0007164644775912166, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00072, + "step": 420, + "tokens/total": 12715008, + "tokens/train_per_sec_per_gpu": 37.25, + "tokens/trainable": 192051 + }, + { + "epoch": 1.64453125, + "grad_norm": 0.01246409397572279, + "learning_rate": 1.7695112982104225e-05, + "loss": 0.0002488879836164415, + "memory/device_reserved (GiB)": 35.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00025, + "step": 421, + "tokens/total": 12745232, + "tokens/train_per_sec_per_gpu": 31.77, + "tokens/trainable": 192500 + }, + { + "epoch": 1.6484375, + "grad_norm": 0.00797255989164114, + "learning_rate": 1.7533544450435433e-05, + "loss": 8.724055078346282e-05, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00009, + "step": 422, + "tokens/total": 12775824, + "tokens/train_per_sec_per_gpu": 33.23, + "tokens/trainable": 192955 + }, + { + "epoch": 1.65234375, + "grad_norm": 0.029961727559566498, + "learning_rate": 1.7373535050913946e-05, + "loss": 0.00036373038892634213, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00036, + "step": 423, + "tokens/total": 12806240, + "tokens/train_per_sec_per_gpu": 38.03, + "tokens/trainable": 193462 + }, + { + "epoch": 1.65625, + "grad_norm": 0.11438705027103424, + "learning_rate": 1.721509144218405e-05, + "loss": 0.0008654020493850112, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00087, + "step": 424, + "tokens/total": 12836512, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 193943 + }, + { + "epoch": 1.66015625, + "grad_norm": 0.11104747653007507, + "learning_rate": 1.705822021773101e-05, + "loss": 0.0019330887589603662, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00193, + "step": 425, + "tokens/total": 12866704, + "tokens/train_per_sec_per_gpu": 29.75, + "tokens/trainable": 194360 + }, + { + "epoch": 1.6640625, + "grad_norm": 0.005441099405288696, + "learning_rate": 1.69029279056068e-05, + "loss": 0.00012030347716063261, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00012, + "step": 426, + "tokens/total": 12896912, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 194848 + }, + { + "epoch": 1.66796875, + "grad_norm": 0.010912942700088024, + "learning_rate": 1.6749220968158415e-05, + "loss": 0.00012806981976609677, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00013, + "step": 427, + "tokens/total": 12927248, + "tokens/train_per_sec_per_gpu": 34.48, + "tokens/trainable": 195314 + }, + { + "epoch": 1.671875, + "grad_norm": 0.007124146446585655, + "learning_rate": 1.659710580175893e-05, + "loss": 9.905424667522311e-05, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0001, + "step": 428, + "tokens/total": 12957632, + "tokens/train_per_sec_per_gpu": 32.4, + "tokens/trainable": 195732 + }, + { + "epoch": 1.67578125, + "grad_norm": 0.23617680370807648, + "learning_rate": 1.644658873654133e-05, + "loss": 0.0024689119309186935, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.00247, + "step": 429, + "tokens/total": 12987824, + "tokens/train_per_sec_per_gpu": 33.37, + "tokens/trainable": 196197 + }, + { + "epoch": 1.6796875, + "grad_norm": 0.29176223278045654, + "learning_rate": 1.629767603613508e-05, + "loss": 0.005283118691295385, + "memory/device_reserved (GiB)": 35.83, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0053, + "step": 430, + "tokens/total": 13018208, + "tokens/train_per_sec_per_gpu": 32.85, + "tokens/trainable": 196678 + }, + { + "epoch": 1.68359375, + "grad_norm": 0.030829036608338356, + "learning_rate": 1.615037389740547e-05, + "loss": 0.0004377455043140799, + "memory/device_reserved (GiB)": 36.36, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.00044, + "step": 431, + "tokens/total": 13048800, + "tokens/train_per_sec_per_gpu": 31.52, + "tokens/trainable": 197109 + }, + { + "epoch": 1.6875, + "grad_norm": 0.023997988551855087, + "learning_rate": 1.600468845019576e-05, + "loss": 0.0002723708748817444, + "memory/device_reserved (GiB)": 36.36, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00027, + "step": 432, + "tokens/total": 13079392, + "tokens/train_per_sec_per_gpu": 39.22, + "tokens/trainable": 197607 + }, + { + "epoch": 1.69140625, + "grad_norm": 0.03983060643076897, + "learning_rate": 1.5860625757072092e-05, + "loss": 0.0006240076618269086, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.00062, + "step": 433, + "tokens/total": 13109920, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 198066 + }, + { + "epoch": 1.6953125, + "grad_norm": 0.03856171295046806, + "learning_rate": 1.571819181307116e-05, + "loss": 0.0005622098105959594, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00056, + "step": 434, + "tokens/total": 13140064, + "tokens/train_per_sec_per_gpu": 33.11, + "tokens/trainable": 198484 + }, + { + "epoch": 1.69921875, + "grad_norm": 0.07453715056180954, + "learning_rate": 1.557739254545075e-05, + "loss": 0.0008548495243303478, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00086, + "step": 435, + "tokens/total": 13170656, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 198957 + }, + { + "epoch": 1.703125, + "grad_norm": 0.0035155205987393856, + "learning_rate": 1.543823381344311e-05, + "loss": 5.2120005420874804e-05, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00005, + "step": 436, + "tokens/total": 13201056, + "tokens/train_per_sec_per_gpu": 33.78, + "tokens/trainable": 199405 + }, + { + "epoch": 1.70703125, + "grad_norm": 0.9568848013877869, + "learning_rate": 1.5300721408011114e-05, + "loss": 0.00043475639540702105, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00043, + "step": 437, + "tokens/total": 13231584, + "tokens/train_per_sec_per_gpu": 34.61, + "tokens/trainable": 199888 + }, + { + "epoch": 1.7109375, + "grad_norm": 0.010019529610872269, + "learning_rate": 1.5164861051607254e-05, + "loss": 0.000111720735731069, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00011, + "step": 438, + "tokens/total": 13261888, + "tokens/train_per_sec_per_gpu": 35.46, + "tokens/trainable": 200396 + }, + { + "epoch": 1.71484375, + "grad_norm": 0.008281780406832695, + "learning_rate": 1.5030658397935521e-05, + "loss": 0.00017071102047339082, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00017, + "step": 439, + "tokens/total": 13291936, + "tokens/train_per_sec_per_gpu": 33.34, + "tokens/trainable": 200843 + }, + { + "epoch": 1.71875, + "grad_norm": 0.06077976152300835, + "learning_rate": 1.4898119031716104e-05, + "loss": 0.0007504730019718409, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00075, + "step": 440, + "tokens/total": 13322224, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 201315 + }, + { + "epoch": 1.72265625, + "grad_norm": 0.12958590686321259, + "learning_rate": 1.476724846845306e-05, + "loss": 0.001083776238374412, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00108, + "step": 441, + "tokens/total": 13352640, + "tokens/train_per_sec_per_gpu": 34.77, + "tokens/trainable": 201782 + }, + { + "epoch": 1.7265625, + "grad_norm": 0.04559873417019844, + "learning_rate": 1.463805215420471e-05, + "loss": 0.0005228023510426283, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.62, + "memory/max_allocated (GiB)": 33.62, + "ppl": 1.00052, + "step": 442, + "tokens/total": 13382560, + "tokens/train_per_sec_per_gpu": 31.27, + "tokens/trainable": 202222 + }, + { + "epoch": 1.73046875, + "grad_norm": 0.02645108290016651, + "learning_rate": 1.451053546535705e-05, + "loss": 0.0003615481255110353, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00036, + "step": 443, + "tokens/total": 13412976, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 202660 + }, + { + "epoch": 1.734375, + "grad_norm": 0.021894006058573723, + "learning_rate": 1.438470370840001e-05, + "loss": 0.000229351528105326, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.64, + "memory/max_allocated (GiB)": 33.64, + "ppl": 1.00023, + "step": 444, + "tokens/total": 13442720, + "tokens/train_per_sec_per_gpu": 31.51, + "tokens/trainable": 203068 + }, + { + "epoch": 1.73828125, + "grad_norm": 0.1398804783821106, + "learning_rate": 1.4260562119706606e-05, + "loss": 0.0013714064843952656, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00137, + "step": 445, + "tokens/total": 13472928, + "tokens/train_per_sec_per_gpu": 34.97, + "tokens/trainable": 203527 + }, + { + "epoch": 1.7421875, + "grad_norm": 0.024557381868362427, + "learning_rate": 1.413811586531508e-05, + "loss": 0.00023480730305891484, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00023, + "step": 446, + "tokens/total": 13502992, + "tokens/train_per_sec_per_gpu": 35.94, + "tokens/trainable": 203994 + }, + { + "epoch": 1.74609375, + "grad_norm": 0.021804476156830788, + "learning_rate": 1.4017370040713884e-05, + "loss": 0.00022578673087991774, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00023, + "step": 447, + "tokens/total": 13533184, + "tokens/train_per_sec_per_gpu": 38.06, + "tokens/trainable": 204430 + }, + { + "epoch": 1.75, + "grad_norm": 0.23265990614891052, + "learning_rate": 1.3898329670629645e-05, + "loss": 0.002107386477291584, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00211, + "step": 448, + "tokens/total": 13563648, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 204886 + }, + { + "epoch": 1.75390625, + "grad_norm": 0.015779122710227966, + "learning_rate": 1.3780999708818058e-05, + "loss": 0.00014222922618500888, + "memory/device_reserved (GiB)": 36.81, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.00014, + "step": 449, + "tokens/total": 13594032, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 205356 + }, + { + "epoch": 1.7578125, + "grad_norm": 0.09757973998785019, + "learning_rate": 1.3665385037857758e-05, + "loss": 0.0012260058429092169, + "memory/device_reserved (GiB)": 35.69, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00123, + "step": 450, + "tokens/total": 13624368, + "tokens/train_per_sec_per_gpu": 31.13, + "tokens/trainable": 205782 + }, + { + "epoch": 1.76171875, + "grad_norm": 0.05971763655543327, + "learning_rate": 1.3551490468947126e-05, + "loss": 0.0006241232040338218, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00062, + "step": 451, + "tokens/total": 13654768, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 206219 + }, + { + "epoch": 1.765625, + "grad_norm": 0.009018263779580593, + "learning_rate": 1.3439320741704075e-05, + "loss": 5.874139242223464e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00006, + "step": 452, + "tokens/total": 13684960, + "tokens/train_per_sec_per_gpu": 36.29, + "tokens/trainable": 206677 + }, + { + "epoch": 1.76953125, + "grad_norm": 0.011312981136143208, + "learning_rate": 1.3328880523968808e-05, + "loss": 8.52083321660757e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00009, + "step": 453, + "tokens/total": 13715488, + "tokens/train_per_sec_per_gpu": 37.8, + "tokens/trainable": 207160 + }, + { + "epoch": 1.7734375, + "grad_norm": 0.0022447824012488127, + "learning_rate": 1.3220174411609587e-05, + "loss": 3.376982203917578e-05, + "memory/device_reserved (GiB)": 35.7, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.00003, + "step": 454, + "tokens/total": 13745856, + "tokens/train_per_sec_per_gpu": 36.25, + "tokens/trainable": 207605 + }, + { + "epoch": 1.77734375, + "grad_norm": 0.017390765249729156, + "learning_rate": 1.3113206928331471e-05, + "loss": 0.00019571901066228747, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0002, + "step": 455, + "tokens/total": 13776192, + "tokens/train_per_sec_per_gpu": 33.43, + "tokens/trainable": 208047 + }, + { + "epoch": 1.78125, + "grad_norm": 0.048852503299713135, + "learning_rate": 1.300798252548806e-05, + "loss": 0.0005775241879746318, + "memory/device_reserved (GiB)": 35.78, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00058, + "step": 456, + "tokens/total": 13806496, + "tokens/train_per_sec_per_gpu": 35.44, + "tokens/trainable": 208476 + }, + { + "epoch": 1.78515625, + "grad_norm": 0.3256935179233551, + "learning_rate": 1.2904505581896265e-05, + "loss": 0.0016588940052315593, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00166, + "step": 457, + "tokens/total": 13836768, + "tokens/train_per_sec_per_gpu": 36.03, + "tokens/trainable": 208970 + }, + { + "epoch": 1.7890625, + "grad_norm": 0.06032567098736763, + "learning_rate": 1.2802780403654082e-05, + "loss": 0.000697342911735177, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.0007, + "step": 458, + "tokens/total": 13867152, + "tokens/train_per_sec_per_gpu": 35.31, + "tokens/trainable": 209456 + }, + { + "epoch": 1.79296875, + "grad_norm": 0.007340370211750269, + "learning_rate": 1.2702811223961408e-05, + "loss": 3.1743944418849424e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00003, + "step": 459, + "tokens/total": 13897360, + "tokens/train_per_sec_per_gpu": 34.15, + "tokens/trainable": 209907 + }, + { + "epoch": 1.796875, + "grad_norm": 0.16073353588581085, + "learning_rate": 1.2604602202943861e-05, + "loss": 0.001599560840986669, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.0016, + "step": 460, + "tokens/total": 13927792, + "tokens/train_per_sec_per_gpu": 34.16, + "tokens/trainable": 210366 + }, + { + "epoch": 1.80078125, + "grad_norm": 0.5971834659576416, + "learning_rate": 1.2508157427479686e-05, + "loss": 0.002240244299173355, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.65, + "memory/max_allocated (GiB)": 33.65, + "ppl": 1.00224, + "step": 461, + "tokens/total": 13957856, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 210835 + }, + { + "epoch": 1.8046875, + "grad_norm": 0.15249626338481903, + "learning_rate": 1.2413480911029655e-05, + "loss": 0.0012240726500749588, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.38, + "memory/max_allocated (GiB)": 33.38, + "ppl": 1.00122, + "step": 462, + "tokens/total": 13986080, + "tokens/train_per_sec_per_gpu": 30.91, + "tokens/trainable": 211259 + }, + { + "epoch": 1.80859375, + "grad_norm": 0.005158649291843176, + "learning_rate": 1.2320576593470082e-05, + "loss": 3.6364894185680896e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.79, + "memory/max_allocated (GiB)": 33.79, + "ppl": 1.00004, + "step": 463, + "tokens/total": 14016288, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 211721 + }, + { + "epoch": 1.8125, + "grad_norm": 0.11264989525079727, + "learning_rate": 1.2229448340928828e-05, + "loss": 0.001997420797124505, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.002, + "step": 464, + "tokens/total": 14046288, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 212159 + }, + { + "epoch": 1.81640625, + "grad_norm": 0.020843395963311195, + "learning_rate": 1.2140099945624458e-05, + "loss": 0.0001486139663029462, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00015, + "step": 465, + "tokens/total": 14076512, + "tokens/train_per_sec_per_gpu": 33.17, + "tokens/trainable": 212644 + }, + { + "epoch": 1.8203125, + "grad_norm": 0.0014335883315652609, + "learning_rate": 1.205253512570841e-05, + "loss": 2.982833029818721e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00003, + "step": 466, + "tokens/total": 14106608, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 213084 + }, + { + "epoch": 1.82421875, + "grad_norm": 0.17584311962127686, + "learning_rate": 1.1966757525110255e-05, + "loss": 0.0020335109438747168, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00204, + "step": 467, + "tokens/total": 14136864, + "tokens/train_per_sec_per_gpu": 32.05, + "tokens/trainable": 213523 + }, + { + "epoch": 1.828125, + "grad_norm": 0.012916951440274715, + "learning_rate": 1.1882770713386095e-05, + "loss": 0.00013626206782646477, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.00014, + "step": 468, + "tokens/total": 14166896, + "tokens/train_per_sec_per_gpu": 35.03, + "tokens/trainable": 213963 + }, + { + "epoch": 1.83203125, + "grad_norm": 0.13297978043556213, + "learning_rate": 1.180057818556998e-05, + "loss": 0.0010875939624384046, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00109, + "step": 469, + "tokens/total": 14197568, + "tokens/train_per_sec_per_gpu": 37.4, + "tokens/trainable": 214431 + }, + { + "epoch": 1.8359375, + "grad_norm": 0.0005420309607870877, + "learning_rate": 1.1720183362028494e-05, + "loss": 1.2743539627990685e-05, + "memory/device_reserved (GiB)": 35.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00001, + "step": 470, + "tokens/total": 14225984, + "tokens/train_per_sec_per_gpu": 36.07, + "tokens/trainable": 214865 + }, + { + "epoch": 1.83984375, + "grad_norm": 0.3001365065574646, + "learning_rate": 1.1641589588318387e-05, + "loss": 0.0015361867845058441, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00154, + "step": 471, + "tokens/total": 14256512, + "tokens/train_per_sec_per_gpu": 35.98, + "tokens/trainable": 215358 + }, + { + "epoch": 1.84375, + "grad_norm": 0.014027237892150879, + "learning_rate": 1.1564800135047418e-05, + "loss": 0.00015228531265165657, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00015, + "step": 472, + "tokens/total": 14286640, + "tokens/train_per_sec_per_gpu": 33.38, + "tokens/trainable": 215800 + }, + { + "epoch": 1.84765625, + "grad_norm": 0.08512959629297256, + "learning_rate": 1.148981819773816e-05, + "loss": 0.000548110285308212, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00055, + "step": 473, + "tokens/total": 14317024, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 216245 + }, + { + "epoch": 1.8515625, + "grad_norm": 0.004200903698801994, + "learning_rate": 1.1416646896695086e-05, + "loss": 4.4981639803154394e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00004, + "step": 474, + "tokens/total": 14347088, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 216685 + }, + { + "epoch": 1.85546875, + "grad_norm": 0.4064914584159851, + "learning_rate": 1.1345289276874717e-05, + "loss": 0.002648166147992015, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00265, + "step": 475, + "tokens/total": 14377152, + "tokens/train_per_sec_per_gpu": 31.31, + "tokens/trainable": 217092 + }, + { + "epoch": 1.859375, + "grad_norm": 0.0013920688070356846, + "learning_rate": 1.1275748307758873e-05, + "loss": 1.866836828412488e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00002, + "step": 476, + "tokens/total": 14407568, + "tokens/train_per_sec_per_gpu": 38.27, + "tokens/trainable": 217582 + }, + { + "epoch": 1.86328125, + "grad_norm": 0.0018797408556565642, + "learning_rate": 1.1208026883231147e-05, + "loss": 2.958982076961547e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00003, + "step": 477, + "tokens/total": 14437792, + "tokens/train_per_sec_per_gpu": 36.2, + "tokens/trainable": 218049 + }, + { + "epoch": 1.8671875, + "grad_norm": 0.10846851021051407, + "learning_rate": 1.1142127821456433e-05, + "loss": 0.000766873883549124, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00077, + "step": 478, + "tokens/total": 14468400, + "tokens/train_per_sec_per_gpu": 33.06, + "tokens/trainable": 218522 + }, + { + "epoch": 1.87109375, + "grad_norm": 0.08651807904243469, + "learning_rate": 1.1078053864763674e-05, + "loss": 0.0007375412969850004, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.00074, + "step": 479, + "tokens/total": 14498416, + "tokens/train_per_sec_per_gpu": 33.32, + "tokens/trainable": 218970 + }, + { + "epoch": 1.875, + "grad_norm": 0.19086171686649323, + "learning_rate": 1.1015807679531756e-05, + "loss": 0.002163141267374158, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.00217, + "step": 480, + "tokens/total": 14528976, + "tokens/train_per_sec_per_gpu": 34.62, + "tokens/trainable": 219411 + }, + { + "epoch": 1.87890625, + "grad_norm": 0.0037188229616731405, + "learning_rate": 1.0955391856078528e-05, + "loss": 3.280482633272186e-05, + "memory/device_reserved (GiB)": 35.92, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.00003, + "step": 481, + "tokens/total": 14558992, + "tokens/train_per_sec_per_gpu": 29.98, + "tokens/trainable": 219837 + }, + { + "epoch": 1.8828125, + "grad_norm": 0.0028978160116821527, + "learning_rate": 1.0896808908553007e-05, + "loss": 2.934904841822572e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00003, + "step": 482, + "tokens/total": 14589488, + "tokens/train_per_sec_per_gpu": 35.61, + "tokens/trainable": 220345 + }, + { + "epoch": 1.88671875, + "grad_norm": 0.030241984874010086, + "learning_rate": 1.0840061274830763e-05, + "loss": 0.0002913741336669773, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.8, + "memory/max_allocated (GiB)": 33.8, + "ppl": 1.00029, + "step": 483, + "tokens/total": 14619584, + "tokens/train_per_sec_per_gpu": 33.58, + "tokens/trainable": 220775 + }, + { + "epoch": 1.890625, + "grad_norm": 0.0025161579251289368, + "learning_rate": 1.0785151316412473e-05, + "loss": 2.7965787012362853e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00003, + "step": 484, + "tokens/total": 14649760, + "tokens/train_per_sec_per_gpu": 35.58, + "tokens/trainable": 221217 + }, + { + "epoch": 1.89453125, + "grad_norm": 0.014441153965890408, + "learning_rate": 1.0732081318325639e-05, + "loss": 8.626213093521073e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00009, + "step": 485, + "tokens/total": 14679920, + "tokens/train_per_sec_per_gpu": 32.07, + "tokens/trainable": 221666 + }, + { + "epoch": 1.8984375, + "grad_norm": 0.016399968415498734, + "learning_rate": 1.0680853489029501e-05, + "loss": 0.00012868674821220338, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00013, + "step": 486, + "tokens/total": 14710432, + "tokens/train_per_sec_per_gpu": 31.95, + "tokens/trainable": 222110 + }, + { + "epoch": 1.90234375, + "grad_norm": 0.021996503695845604, + "learning_rate": 1.0631469960323152e-05, + "loss": 0.00015319210069719702, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.67, + "memory/max_allocated (GiB)": 33.67, + "ppl": 1.00015, + "step": 487, + "tokens/total": 14740416, + "tokens/train_per_sec_per_gpu": 32.47, + "tokens/trainable": 222542 + }, + { + "epoch": 1.90625, + "grad_norm": 0.002610082970932126, + "learning_rate": 1.0583932787256783e-05, + "loss": 1.0041851055575535e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00001, + "step": 488, + "tokens/total": 14770832, + "tokens/train_per_sec_per_gpu": 36.81, + "tokens/trainable": 223035 + }, + { + "epoch": 1.91015625, + "grad_norm": 0.0016527038533240557, + "learning_rate": 1.0538243948046206e-05, + "loss": 1.9269999029347673e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.00002, + "step": 489, + "tokens/total": 14801104, + "tokens/train_per_sec_per_gpu": 32.27, + "tokens/trainable": 223488 + }, + { + "epoch": 1.9140625, + "grad_norm": 0.0007155581843107939, + "learning_rate": 1.0494405343990523e-05, + "loss": 6.305221177171916e-06, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.00001, + "step": 490, + "tokens/total": 14831216, + "tokens/train_per_sec_per_gpu": 29.86, + "tokens/trainable": 223932 + }, + { + "epoch": 1.91796875, + "grad_norm": 0.013574068434536457, + "learning_rate": 1.0452418799392985e-05, + "loss": 0.00013808548101224005, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00014, + "step": 491, + "tokens/total": 14861488, + "tokens/train_per_sec_per_gpu": 33.39, + "tokens/trainable": 224389 + }, + { + "epoch": 1.921875, + "grad_norm": 0.21587051451206207, + "learning_rate": 1.0412286061485102e-05, + "loss": 0.001721705892123282, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.01, + "memory/max_allocated (GiB)": 34.01, + "ppl": 1.00172, + "step": 492, + "tokens/total": 14891824, + "tokens/train_per_sec_per_gpu": 35.65, + "tokens/trainable": 224871 + }, + { + "epoch": 1.92578125, + "grad_norm": 0.4014979600906372, + "learning_rate": 1.03740088003539e-05, + "loss": 0.016092870384454727, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01622, + "step": 493, + "tokens/total": 14922192, + "tokens/train_per_sec_per_gpu": 35.16, + "tokens/trainable": 225352 + }, + { + "epoch": 1.9296875, + "grad_norm": 0.0035919942893087864, + "learning_rate": 1.0337588608872463e-05, + "loss": 5.178125138627365e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.00005, + "step": 494, + "tokens/total": 14952624, + "tokens/train_per_sec_per_gpu": 33.3, + "tokens/trainable": 225801 + }, + { + "epoch": 1.93359375, + "grad_norm": 0.0015177514869719744, + "learning_rate": 1.0303027002633622e-05, + "loss": 1.5969462765497155e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00002, + "step": 495, + "tokens/total": 14982704, + "tokens/train_per_sec_per_gpu": 33.05, + "tokens/trainable": 226240 + }, + { + "epoch": 1.9375, + "grad_norm": 0.0016166599234566092, + "learning_rate": 1.0270325419886884e-05, + "loss": 2.797747583827004e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.00003, + "step": 496, + "tokens/total": 15013024, + "tokens/train_per_sec_per_gpu": 35.32, + "tokens/trainable": 226712 + }, + { + "epoch": 1.94140625, + "grad_norm": 0.007683792617172003, + "learning_rate": 1.0239485221478599e-05, + "loss": 9.649172716308385e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.0001, + "step": 497, + "tokens/total": 15043520, + "tokens/train_per_sec_per_gpu": 32.28, + "tokens/trainable": 227135 + }, + { + "epoch": 1.9453125, + "grad_norm": 0.0021688074339181185, + "learning_rate": 1.0210507690795292e-05, + "loss": 3.7211153539828956e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.00004, + "step": 498, + "tokens/total": 15073824, + "tokens/train_per_sec_per_gpu": 38.25, + "tokens/trainable": 227600 + }, + { + "epoch": 1.94921875, + "grad_norm": 0.028605012223124504, + "learning_rate": 1.0183394033710305e-05, + "loss": 0.0003017223789356649, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.0003, + "step": 499, + "tokens/total": 15104128, + "tokens/train_per_sec_per_gpu": 37.21, + "tokens/trainable": 228076 + }, + { + "epoch": 1.953125, + "grad_norm": 0.020242059603333473, + "learning_rate": 1.0158145378533583e-05, + "loss": 6.54559480608441e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.00007, + "step": 500, + "tokens/total": 15134480, + "tokens/train_per_sec_per_gpu": 31.41, + "tokens/trainable": 228526 + }, + { + "epoch": 1.95703125, + "grad_norm": 0.0019631576724350452, + "learning_rate": 1.0134762775964726e-05, + "loss": 2.9010820071562193e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00003, + "step": 501, + "tokens/total": 15164832, + "tokens/train_per_sec_per_gpu": 35.76, + "tokens/trainable": 229005 + }, + { + "epoch": 1.9609375, + "grad_norm": 0.05687836930155754, + "learning_rate": 1.0113247199049278e-05, + "loss": 0.0003191005380358547, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.03, + "memory/max_allocated (GiB)": 34.03, + "ppl": 1.00032, + "step": 502, + "tokens/total": 15195616, + "tokens/train_per_sec_per_gpu": 31.5, + "tokens/trainable": 229438 + }, + { + "epoch": 1.96484375, + "grad_norm": 0.04075845330953598, + "learning_rate": 1.0093599543138205e-05, + "loss": 0.00027428093017078936, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00027, + "step": 503, + "tokens/total": 15226096, + "tokens/train_per_sec_per_gpu": 33.33, + "tokens/trainable": 229885 + }, + { + "epoch": 1.96875, + "grad_norm": 0.01305411383509636, + "learning_rate": 1.0075820625850675e-05, + "loss": 0.00012350856559351087, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.00012, + "step": 504, + "tokens/total": 15256512, + "tokens/train_per_sec_per_gpu": 39.53, + "tokens/trainable": 230389 + }, + { + "epoch": 1.97265625, + "grad_norm": 0.08244533091783524, + "learning_rate": 1.0059911187040013e-05, + "loss": 0.0005182232707738876, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.00052, + "step": 505, + "tokens/total": 15287072, + "tokens/train_per_sec_per_gpu": 38.08, + "tokens/trainable": 230885 + }, + { + "epoch": 1.9765625, + "grad_norm": 0.001344834454357624, + "learning_rate": 1.0045871888762893e-05, + "loss": 1.9871291442541406e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00002, + "step": 506, + "tokens/total": 15317296, + "tokens/train_per_sec_per_gpu": 34.79, + "tokens/trainable": 231335 + }, + { + "epoch": 1.98046875, + "grad_norm": 0.0459468699991703, + "learning_rate": 1.003370331525184e-05, + "loss": 0.00030574080301448703, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00031, + "step": 507, + "tokens/total": 15347904, + "tokens/train_per_sec_per_gpu": 36.96, + "tokens/trainable": 231824 + }, + { + "epoch": 1.984375, + "grad_norm": 0.0023596102837473154, + "learning_rate": 1.002340597289085e-05, + "loss": 2.902157575590536e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.33, + "memory/max_allocated (GiB)": 33.33, + "ppl": 1.00003, + "step": 508, + "tokens/total": 15376144, + "tokens/train_per_sec_per_gpu": 32.08, + "tokens/trainable": 232241 + }, + { + "epoch": 1.98828125, + "grad_norm": 0.09540333598852158, + "learning_rate": 1.0014980290194387e-05, + "loss": 0.0011963924625888467, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.0012, + "step": 509, + "tokens/total": 15406976, + "tokens/train_per_sec_per_gpu": 37.83, + "tokens/trainable": 232742 + }, + { + "epoch": 1.9921875, + "grad_norm": 0.0034284063149243593, + "learning_rate": 1.0008426617789489e-05, + "loss": 4.3011394154746085e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00004, + "step": 510, + "tokens/total": 15437344, + "tokens/train_per_sec_per_gpu": 36.66, + "tokens/trainable": 233231 + }, + { + "epoch": 1.99609375, + "grad_norm": 0.01952311210334301, + "learning_rate": 1.0003745228401215e-05, + "loss": 0.0001366769429296255, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.00014, + "step": 511, + "tokens/total": 15467712, + "tokens/train_per_sec_per_gpu": 30.86, + "tokens/trainable": 233655 + }, + { + "epoch": 2.0, + "grad_norm": 0.004961141850799322, + "learning_rate": 1.0000936316841296e-05, + "loss": 8.935239748097956e-05, + "memory/device_reserved (GiB)": 36.15, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.00009, + "step": 512, + "tokens/total": 15498240, + "tokens/train_per_sec_per_gpu": 36.78, + "tokens/trainable": 234124 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": true + }, + "attributes": {} + } + }, + "total_flos": 1.0519561165470106e+18, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-512/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..eabf2fc0075e05ccaa1661dfe24448bb2f5275aa --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:405f4b9c343f437b8c4364d7212cae510d9d8adbf2c91918e11fbd8eaa766321 +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..e4f0443306bd6632fd203c044bd11f885a66ec17 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e57e479684c5374f4a3e01aa49e216786c3b13dffcf2843772779a92d66d5d4b +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..f7b594b334e7aa0fb626dd09f5ad390cd2c21a51 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0c1c39dd1f41c6a2efef09bbfbb45aea5f6c893008225984765de40d9ce68a16 +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..34b7fc400f1006f136b1c89d6c58cf3d17830d72 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15b31b2361cee4c0a1206aa5d9efeb71d8dc96ceaff5b2fe054baf0716df3503 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..69da360b22bd76b0a34e42898e12dfd0a8a86107 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/tokens_state.json @@ -0,0 +1 @@ +{"total": 1940608, "trainable": 29222} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..1c76888b045e11459c2846b4ee5d0c8b3a833b71 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/trainer_state.json @@ -0,0 +1,930 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.25, + "eval_steps": 500, + "global_step": 64, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.3172040537635635e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-64/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/README.md b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/README.md new file mode 100644 index 0000000000000000000000000000000000000000..92a2f21dad9bc1659dbe16a6289991ead973e65b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/README.md @@ -0,0 +1,208 @@ +--- +base_model: /workspace/wave/parent +library_name: peft +pipeline_tag: text-generation +tags: +- axolotl +- base_model:adapter:/workspace/wave/parent +- lora +- transformers +--- + +# Model Card for Model ID + + + + + +## Model Details + +### Model Description + + + + + +- **Developed by:** [More Information Needed] +- **Funded by [optional]:** [More Information Needed] +- **Shared by [optional]:** [More Information Needed] +- **Model type:** [More Information Needed] +- **Language(s) (NLP):** [More Information Needed] +- **License:** [More Information Needed] +- **Finetuned from model [optional]:** [More Information Needed] + +### Model Sources [optional] + + + +- **Repository:** [More Information Needed] +- **Paper [optional]:** [More Information Needed] +- **Demo [optional]:** [More Information Needed] + +## Uses + + + +### Direct Use + + + +[More Information Needed] + +### Downstream Use [optional] + + + +[More Information Needed] + +### Out-of-Scope Use + + + +[More Information Needed] + +## Bias, Risks, and Limitations + + + +[More Information Needed] + +### Recommendations + + + +Users (both direct and downstream) should be made aware of the risks, biases and limitations of the model. More information needed for further recommendations. + +## How to Get Started with the Model + +Use the code below to get started with the model. + +[More Information Needed] + +## Training Details + +### Training Data + + + +[More Information Needed] + +### Training Procedure + + + +#### Preprocessing [optional] + +[More Information Needed] + + +#### Training Hyperparameters + +- **Training regime:** [More Information Needed] + +#### Speeds, Sizes, Times [optional] + + + +[More Information Needed] + +## Evaluation + + + +### Testing Data, Factors & Metrics + +#### Testing Data + + + +[More Information Needed] + +#### Factors + + + +[More Information Needed] + +#### Metrics + + + +[More Information Needed] + +### Results + +[More Information Needed] + +#### Summary + + + +## Model Examination [optional] + + + +[More Information Needed] + +## Environmental Impact + + + +Carbon emissions can be estimated using the [Machine Learning Impact calculator](https://mlco2.github.io/impact#compute) presented in [Lacoste et al. (2019)](https://arxiv.org/abs/1910.09700). + +- **Hardware Type:** [More Information Needed] +- **Hours used:** [More Information Needed] +- **Cloud Provider:** [More Information Needed] +- **Compute Region:** [More Information Needed] +- **Carbon Emitted:** [More Information Needed] + +## Technical Specifications [optional] + +### Model Architecture and Objective + +[More Information Needed] + +### Compute Infrastructure + +[More Information Needed] + +#### Hardware + +[More Information Needed] + +#### Software + +[More Information Needed] + +## Citation [optional] + + + +**BibTeX:** + +[More Information Needed] + +**APA:** + +[More Information Needed] + +## Glossary [optional] + + + +[More Information Needed] + +## More Information [optional] + +[More Information Needed] + +## Model Card Authors [optional] + +[More Information Needed] + +## Model Card Contact + +[More Information Needed] +### Framework versions + +- PEFT 0.19.1 \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/adapter_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/adapter_config.json new file mode 100644 index 0000000000000000000000000000000000000000..9503a396dadc7378b377d1fd44bcb6c04a11922e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/adapter_config.json @@ -0,0 +1,48 @@ +{ + "alora_invocation_tokens": null, + "alpha_pattern": {}, + "arrow_config": null, + "auto_mapping": null, + "base_model_name_or_path": "/workspace/wave/parent", + "bias": "none", + "corda_config": null, + "ensure_weight_tying": false, + "eva_config": null, + "exclude_modules": null, + "fan_in_fan_out": null, + "inference_mode": true, + "init_lora_weights": true, + "layer_replication": null, + "layers_pattern": null, + "layers_to_transform": null, + "loftq_config": {}, + "lora_alpha": 64, + "lora_bias": false, + "lora_dropout": 0.05, + "lora_ga_config": null, + "megatron_config": null, + "megatron_core": "megatron.core", + "modules_to_save": null, + "peft_type": "LORA", + "peft_version": "0.19.1", + "qalora_group_size": 16, + "r": 32, + "rank_pattern": {}, + "revision": null, + "target_modules": [ + "o_proj", + "q_proj", + "k_proj", + "down_proj", + "gate_proj", + "v_proj", + "up_proj" + ], + "target_parameters": [], + "task_type": "CAUSAL_LM", + "trainable_token_indices": null, + "use_bdlora": null, + "use_dora": false, + "use_qalora": false, + "use_rslora": false +} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/adapter_model.safetensors b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/adapter_model.safetensors new file mode 100644 index 0000000000000000000000000000000000000000..40f2097285680843ccf86793068c3f3af0437df7 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/adapter_model.safetensors @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ee64598865679ab0749385b9a85d0fa758de86dfe72510c7116064aef45ddede +size 547777976 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/chat_template.jinja b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/chat_template.jinja new file mode 100644 index 0000000000000000000000000000000000000000..1117055ab8e8c90e1b200be00cddb78943616d9e --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/chat_template.jinja @@ -0,0 +1,47 @@ +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/optimizer.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/optimizer.pt new file mode 100644 index 0000000000000000000000000000000000000000..966f1f1f5addad9d0ec80cce1d69c5580e11c5c9 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/optimizer.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2e0ff534f3d7cfd3ac1b559bfc3cd0ec2b5785000abac241e533c30e495f0721 +size 1048106435 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/rng_state.pth b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/rng_state.pth new file mode 100644 index 0000000000000000000000000000000000000000..efa3b1611bbfcc2c397824e4bcd225be7a55c283 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/rng_state.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:adbeb73951e9ff069c0226d3d5550f1fe22c83b0c416ebc3684c1f8fff3f48ce +size 14645 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/scheduler.pt b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/scheduler.pt new file mode 100644 index 0000000000000000000000000000000000000000..433f82b14836de77d366e2c961bc5bdacdc9eb63 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/scheduler.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:786444fedf73fac372c74a0ffd25119b2bb7107a3f00f8bfd9a8a46a6f2f4ff0 +size 1465 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..f74d183147799f3fd56768db82b27473f0e09a1d --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer.json @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:daab2354f8a74e70d70b4d1f804939b68a8c9624dd06cb7858e52dd8970e9726 +size 33384567 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer_config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer_config.json new file mode 100644 index 0000000000000000000000000000000000000000..ec6b4167126c13f59aef9ea855c12284a2a86619 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokenizer_config.json @@ -0,0 +1,26 @@ +{ + "backend": "tokenizers", + "boi_token": "", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokens_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokens_state.json new file mode 100644 index 0000000000000000000000000000000000000000..6316bb77a252c8b4462ed250bd46b316485f81f8 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/tokens_state.json @@ -0,0 +1 @@ +{"total": 2906096, "trainable": 43869} \ No newline at end of file diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/trainer_state.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..274841fe8972c7f3fa53acc878a63d46fd61d6ef --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/trainer_state.json @@ -0,0 +1,1378 @@ +{ + "best_global_step": null, + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 0.375, + "eval_steps": 500, + "global_step": 96, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.00390625, + "grad_norm": 8.088534355163574, + "learning_rate": 0.0, + "loss": 0.1503293514251709, + "memory/device_reserved (GiB)": 33.9, + "memory/max_active (GiB)": 32.79, + "memory/max_allocated (GiB)": 32.79, + "ppl": 1.16222, + "step": 1, + "tokens/total": 30176, + "tokens/train_per_sec_per_gpu": 28.32, + "tokens/trainable": 425 + }, + { + "epoch": 0.0078125, + "grad_norm": 0.8953529596328735, + "learning_rate": 4.000000000000001e-06, + "loss": 0.15210241079330444, + "memory/device_reserved (GiB)": 34.68, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.16428, + "step": 2, + "tokens/total": 60272, + "tokens/train_per_sec_per_gpu": 33.91, + "tokens/trainable": 872 + }, + { + "epoch": 0.01171875, + "grad_norm": 1.4013757705688477, + "learning_rate": 8.000000000000001e-06, + "loss": 0.14530743658542633, + "memory/device_reserved (GiB)": 34.72, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.1564, + "step": 3, + "tokens/total": 90512, + "tokens/train_per_sec_per_gpu": 35.5, + "tokens/trainable": 1352 + }, + { + "epoch": 0.015625, + "grad_norm": 0.9737329483032227, + "learning_rate": 1.2e-05, + "loss": 0.13504940271377563, + "memory/device_reserved (GiB)": 34.73, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.14459, + "step": 4, + "tokens/total": 120944, + "tokens/train_per_sec_per_gpu": 31.75, + "tokens/trainable": 1777 + }, + { + "epoch": 0.01953125, + "grad_norm": 1.5541704893112183, + "learning_rate": 1.6000000000000003e-05, + "loss": 0.13814105093479156, + "memory/device_reserved (GiB)": 35.12, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.14814, + "step": 5, + "tokens/total": 151440, + "tokens/train_per_sec_per_gpu": 37.35, + "tokens/trainable": 2260 + }, + { + "epoch": 0.0234375, + "grad_norm": 6.433568477630615, + "learning_rate": 2e-05, + "loss": 0.12622183561325073, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.13453, + "step": 6, + "tokens/total": 181984, + "tokens/train_per_sec_per_gpu": 33.21, + "tokens/trainable": 2704 + }, + { + "epoch": 0.02734375, + "grad_norm": 1.4011743068695068, + "learning_rate": 2.4e-05, + "loss": 0.11199458688497543, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.11851, + "step": 7, + "tokens/total": 212336, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 3143 + }, + { + "epoch": 0.03125, + "grad_norm": 2.038933038711548, + "learning_rate": 2.8000000000000003e-05, + "loss": 0.09219667315483093, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.09658, + "step": 8, + "tokens/total": 242592, + "tokens/train_per_sec_per_gpu": 34.78, + "tokens/trainable": 3586 + }, + { + "epoch": 0.03515625, + "grad_norm": 4.82806396484375, + "learning_rate": 3.2000000000000005e-05, + "loss": 0.04823308065533638, + "memory/device_reserved (GiB)": 36.12, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04942, + "step": 9, + "tokens/total": 272784, + "tokens/train_per_sec_per_gpu": 29.59, + "tokens/trainable": 4007 + }, + { + "epoch": 0.0390625, + "grad_norm": 1.3691715002059937, + "learning_rate": 3.6e-05, + "loss": 0.04049058258533478, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.04132, + "step": 10, + "tokens/total": 303184, + "tokens/train_per_sec_per_gpu": 39.66, + "tokens/trainable": 4481 + }, + { + "epoch": 0.04296875, + "grad_norm": 2.285440444946289, + "learning_rate": 4e-05, + "loss": 0.06339800357818604, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.06545, + "step": 11, + "tokens/total": 333296, + "tokens/train_per_sec_per_gpu": 35.09, + "tokens/trainable": 4958 + }, + { + "epoch": 0.046875, + "grad_norm": 5.0530476570129395, + "learning_rate": 4.4000000000000006e-05, + "loss": 0.07100467383861542, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.07359, + "step": 12, + "tokens/total": 363840, + "tokens/train_per_sec_per_gpu": 34.81, + "tokens/trainable": 5440 + }, + { + "epoch": 0.05078125, + "grad_norm": 3.5007340908050537, + "learning_rate": 4.8e-05, + "loss": 0.061270229518413544, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06319, + "step": 13, + "tokens/total": 394112, + "tokens/train_per_sec_per_gpu": 35.82, + "tokens/trainable": 5871 + }, + { + "epoch": 0.0546875, + "grad_norm": 2.6390483379364014, + "learning_rate": 5.2000000000000004e-05, + "loss": 0.043673258274793625, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.04464, + "step": 14, + "tokens/total": 424656, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 6360 + }, + { + "epoch": 0.05859375, + "grad_norm": 2.6278514862060547, + "learning_rate": 5.6000000000000006e-05, + "loss": 0.06025860086083412, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.96, + "memory/max_allocated (GiB)": 33.96, + "ppl": 1.06211, + "step": 15, + "tokens/total": 455152, + "tokens/train_per_sec_per_gpu": 36.67, + "tokens/trainable": 6841 + }, + { + "epoch": 0.0625, + "grad_norm": 5.878053665161133, + "learning_rate": 6e-05, + "loss": 0.09122475981712341, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.09552, + "step": 16, + "tokens/total": 485328, + "tokens/train_per_sec_per_gpu": 31.7, + "tokens/trainable": 7254 + }, + { + "epoch": 0.06640625, + "grad_norm": 1.9717961549758911, + "learning_rate": 6.400000000000001e-05, + "loss": 0.017721600830554962, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01788, + "step": 17, + "tokens/total": 515744, + "tokens/train_per_sec_per_gpu": 36.52, + "tokens/trainable": 7723 + }, + { + "epoch": 0.0703125, + "grad_norm": 1.783061146736145, + "learning_rate": 6.800000000000001e-05, + "loss": 0.04740295931696892, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04854, + "step": 18, + "tokens/total": 546144, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 8208 + }, + { + "epoch": 0.07421875, + "grad_norm": 1.916291356086731, + "learning_rate": 7.2e-05, + "loss": 0.025028439238667488, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02534, + "step": 19, + "tokens/total": 576560, + "tokens/train_per_sec_per_gpu": 35.49, + "tokens/trainable": 8699 + }, + { + "epoch": 0.078125, + "grad_norm": 2.9791574478149414, + "learning_rate": 7.6e-05, + "loss": 0.055660396814346313, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.05724, + "step": 20, + "tokens/total": 607056, + "tokens/train_per_sec_per_gpu": 40.56, + "tokens/trainable": 9214 + }, + { + "epoch": 0.08203125, + "grad_norm": 1.2719122171401978, + "learning_rate": 8e-05, + "loss": 0.024366460740566254, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.02467, + "step": 21, + "tokens/total": 637488, + "tokens/train_per_sec_per_gpu": 32.12, + "tokens/trainable": 9639 + }, + { + "epoch": 0.0859375, + "grad_norm": 2.241725444793701, + "learning_rate": 8.4e-05, + "loss": 0.05170319974422455, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.75, + "memory/max_allocated (GiB)": 33.75, + "ppl": 1.05306, + "step": 22, + "tokens/total": 667632, + "tokens/train_per_sec_per_gpu": 36.71, + "tokens/trainable": 10111 + }, + { + "epoch": 0.08984375, + "grad_norm": 2.2158987522125244, + "learning_rate": 8.800000000000001e-05, + "loss": 0.04347127676010132, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.04443, + "step": 23, + "tokens/total": 697872, + "tokens/train_per_sec_per_gpu": 33.4, + "tokens/trainable": 10560 + }, + { + "epoch": 0.09375, + "grad_norm": 1.028172254562378, + "learning_rate": 9.200000000000001e-05, + "loss": 0.016212783753871918, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01634, + "step": 24, + "tokens/total": 728432, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 11018 + }, + { + "epoch": 0.09765625, + "grad_norm": 1.0408563613891602, + "learning_rate": 9.6e-05, + "loss": 0.023505806922912598, + "memory/device_reserved (GiB)": 37.81, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02378, + "step": 25, + "tokens/total": 756656, + "tokens/train_per_sec_per_gpu": 40.53, + "tokens/trainable": 11442 + }, + { + "epoch": 0.1015625, + "grad_norm": 1.6656584739685059, + "learning_rate": 0.0001, + "loss": 0.05568547546863556, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.05727, + "step": 26, + "tokens/total": 787120, + "tokens/train_per_sec_per_gpu": 34.83, + "tokens/trainable": 11899 + }, + { + "epoch": 0.10546875, + "grad_norm": 0.7106256484985352, + "learning_rate": 9.99990636831587e-05, + "loss": 0.009144511073827744, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.00919, + "step": 27, + "tokens/total": 817504, + "tokens/train_per_sec_per_gpu": 32.1, + "tokens/trainable": 12341 + }, + { + "epoch": 0.109375, + "grad_norm": 0.8463047742843628, + "learning_rate": 9.999625477159879e-05, + "loss": 0.020250165835022926, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.02046, + "step": 28, + "tokens/total": 848128, + "tokens/train_per_sec_per_gpu": 38.39, + "tokens/trainable": 12825 + }, + { + "epoch": 0.11328125, + "grad_norm": 1.0744967460632324, + "learning_rate": 9.999157338221051e-05, + "loss": 0.023447610437870026, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02372, + "step": 29, + "tokens/total": 878448, + "tokens/train_per_sec_per_gpu": 36.01, + "tokens/trainable": 13295 + }, + { + "epoch": 0.1171875, + "grad_norm": 1.4234495162963867, + "learning_rate": 9.998501970980562e-05, + "loss": 0.03458714857697487, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.94, + "memory/max_allocated (GiB)": 33.94, + "ppl": 1.03519, + "step": 30, + "tokens/total": 909008, + "tokens/train_per_sec_per_gpu": 35.8, + "tokens/trainable": 13784 + }, + { + "epoch": 0.12109375, + "grad_norm": 0.9880445003509521, + "learning_rate": 9.997659402710915e-05, + "loss": 0.02479860559105873, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02511, + "step": 31, + "tokens/total": 939312, + "tokens/train_per_sec_per_gpu": 34.58, + "tokens/trainable": 14250 + }, + { + "epoch": 0.125, + "grad_norm": 1.1164203882217407, + "learning_rate": 9.996629668474818e-05, + "loss": 0.023022079840302467, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02329, + "step": 32, + "tokens/total": 969264, + "tokens/train_per_sec_per_gpu": 35.0, + "tokens/trainable": 14680 + }, + { + "epoch": 0.12890625, + "grad_norm": 1.324739694595337, + "learning_rate": 9.995412811123711e-05, + "loss": 0.0630958303809166, + "memory/device_reserved (GiB)": 37.82, + "memory/max_active (GiB)": 33.95, + "memory/max_allocated (GiB)": 33.95, + "ppl": 1.06513, + "step": 33, + "tokens/total": 999792, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 15112 + }, + { + "epoch": 0.1328125, + "grad_norm": 1.0034446716308594, + "learning_rate": 9.994008881295999e-05, + "loss": 0.04252947121858597, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.04345, + "step": 34, + "tokens/total": 1029840, + "tokens/train_per_sec_per_gpu": 34.46, + "tokens/trainable": 15573 + }, + { + "epoch": 0.13671875, + "grad_norm": 1.1391361951828003, + "learning_rate": 9.992417937414932e-05, + "loss": 0.0361175499856472, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.03678, + "step": 35, + "tokens/total": 1060416, + "tokens/train_per_sec_per_gpu": 34.24, + "tokens/trainable": 16062 + }, + { + "epoch": 0.140625, + "grad_norm": 0.48655080795288086, + "learning_rate": 9.99064004568618e-05, + "loss": 0.015174772590398788, + "memory/device_reserved (GiB)": 35.48, + "memory/max_active (GiB)": 33.72, + "memory/max_allocated (GiB)": 33.72, + "ppl": 1.01529, + "step": 36, + "tokens/total": 1090464, + "tokens/train_per_sec_per_gpu": 31.53, + "tokens/trainable": 16500 + }, + { + "epoch": 0.14453125, + "grad_norm": 0.7631387114524841, + "learning_rate": 9.988675280095074e-05, + "loss": 0.04621092230081558, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.0473, + "step": 37, + "tokens/total": 1120944, + "tokens/train_per_sec_per_gpu": 34.9, + "tokens/trainable": 16936 + }, + { + "epoch": 0.1484375, + "grad_norm": 1.1162818670272827, + "learning_rate": 9.986523722403528e-05, + "loss": 0.046002037823200226, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.04708, + "step": 38, + "tokens/total": 1151360, + "tokens/train_per_sec_per_gpu": 29.32, + "tokens/trainable": 17357 + }, + { + "epoch": 0.15234375, + "grad_norm": 4.360172271728516, + "learning_rate": 9.984185462146642e-05, + "loss": 0.052004992961883545, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.05338, + "step": 39, + "tokens/total": 1181728, + "tokens/train_per_sec_per_gpu": 36.82, + "tokens/trainable": 17831 + }, + { + "epoch": 0.15625, + "grad_norm": 0.8433209657669067, + "learning_rate": 9.98166059662897e-05, + "loss": 0.04868919029831886, + "memory/device_reserved (GiB)": 35.86, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.04989, + "step": 40, + "tokens/total": 1212064, + "tokens/train_per_sec_per_gpu": 36.15, + "tokens/trainable": 18315 + }, + { + "epoch": 0.16015625, + "grad_norm": 0.9528040289878845, + "learning_rate": 9.978949230920472e-05, + "loss": 0.029145417734980583, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.02957, + "step": 41, + "tokens/total": 1242592, + "tokens/train_per_sec_per_gpu": 32.49, + "tokens/trainable": 18772 + }, + { + "epoch": 0.1640625, + "grad_norm": 0.5701055526733398, + "learning_rate": 9.976051477852141e-05, + "loss": 0.01586967147886753, + "memory/device_reserved (GiB)": 35.88, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.016, + "step": 42, + "tokens/total": 1272800, + "tokens/train_per_sec_per_gpu": 32.32, + "tokens/trainable": 19197 + }, + { + "epoch": 0.16796875, + "grad_norm": 0.372957706451416, + "learning_rate": 9.972967458011312e-05, + "loss": 0.01052568107843399, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 34.04, + "memory/max_allocated (GiB)": 34.04, + "ppl": 1.01058, + "step": 43, + "tokens/total": 1303600, + "tokens/train_per_sec_per_gpu": 31.68, + "tokens/trainable": 19647 + }, + { + "epoch": 0.171875, + "grad_norm": 0.550298810005188, + "learning_rate": 9.96969729973664e-05, + "loss": 0.019961029291152954, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.02016, + "step": 44, + "tokens/total": 1333936, + "tokens/train_per_sec_per_gpu": 33.83, + "tokens/trainable": 20097 + }, + { + "epoch": 0.17578125, + "grad_norm": 3.7621572017669678, + "learning_rate": 9.966241139112754e-05, + "loss": 0.06465393304824829, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.06679, + "step": 45, + "tokens/total": 1364288, + "tokens/train_per_sec_per_gpu": 31.92, + "tokens/trainable": 20529 + }, + { + "epoch": 0.1796875, + "grad_norm": 1.1490390300750732, + "learning_rate": 9.96259911996461e-05, + "loss": 0.055192310363054276, + "memory/device_reserved (GiB)": 36.04, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.05674, + "step": 46, + "tokens/total": 1394432, + "tokens/train_per_sec_per_gpu": 34.68, + "tokens/trainable": 20973 + }, + { + "epoch": 0.18359375, + "grad_norm": 0.5583573579788208, + "learning_rate": 9.958771393851491e-05, + "loss": 0.024018850177526474, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.98, + "memory/max_allocated (GiB)": 33.98, + "ppl": 1.02431, + "step": 47, + "tokens/total": 1425088, + "tokens/train_per_sec_per_gpu": 35.36, + "tokens/trainable": 21437 + }, + { + "epoch": 0.1875, + "grad_norm": 0.7726427912712097, + "learning_rate": 9.954758120060702e-05, + "loss": 0.021447816863656044, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.02168, + "step": 48, + "tokens/total": 1455456, + "tokens/train_per_sec_per_gpu": 35.57, + "tokens/trainable": 21860 + }, + { + "epoch": 0.19140625, + "grad_norm": 0.9507198929786682, + "learning_rate": 9.950559465600948e-05, + "loss": 0.03307538479566574, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.7, + "memory/max_allocated (GiB)": 33.7, + "ppl": 1.03363, + "step": 49, + "tokens/total": 1485376, + "tokens/train_per_sec_per_gpu": 30.1, + "tokens/trainable": 22273 + }, + { + "epoch": 0.1953125, + "grad_norm": 0.5201236009597778, + "learning_rate": 9.946175605195379e-05, + "loss": 0.019475962966680527, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01967, + "step": 50, + "tokens/total": 1515744, + "tokens/train_per_sec_per_gpu": 36.83, + "tokens/trainable": 22720 + }, + { + "epoch": 0.19921875, + "grad_norm": 0.5436747670173645, + "learning_rate": 9.941606721274322e-05, + "loss": 0.021235385909676552, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02146, + "step": 51, + "tokens/total": 1545984, + "tokens/train_per_sec_per_gpu": 32.55, + "tokens/trainable": 23168 + }, + { + "epoch": 0.203125, + "grad_norm": 0.47054970264434814, + "learning_rate": 9.936853003967685e-05, + "loss": 0.02223706617951393, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02249, + "step": 52, + "tokens/total": 1576288, + "tokens/train_per_sec_per_gpu": 37.44, + "tokens/trainable": 23663 + }, + { + "epoch": 0.20703125, + "grad_norm": 12.063824653625488, + "learning_rate": 9.93191465109705e-05, + "loss": 0.01991445943713188, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02011, + "step": 53, + "tokens/total": 1606528, + "tokens/train_per_sec_per_gpu": 34.49, + "tokens/trainable": 24128 + }, + { + "epoch": 0.2109375, + "grad_norm": 0.731513500213623, + "learning_rate": 9.926791868167438e-05, + "loss": 0.035772789269685745, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.66, + "memory/max_allocated (GiB)": 33.66, + "ppl": 1.03642, + "step": 54, + "tokens/total": 1636400, + "tokens/train_per_sec_per_gpu": 38.33, + "tokens/trainable": 24601 + }, + { + "epoch": 0.21484375, + "grad_norm": 0.8308107256889343, + "learning_rate": 9.921484868358753e-05, + "loss": 0.04387127608060837, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.04485, + "step": 55, + "tokens/total": 1666896, + "tokens/train_per_sec_per_gpu": 34.44, + "tokens/trainable": 25099 + }, + { + "epoch": 0.21875, + "grad_norm": 0.3609774708747864, + "learning_rate": 9.915993872516924e-05, + "loss": 0.009335450828075409, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.00938, + "step": 56, + "tokens/total": 1697232, + "tokens/train_per_sec_per_gpu": 37.08, + "tokens/trainable": 25573 + }, + { + "epoch": 0.22265625, + "grad_norm": 0.5462154150009155, + "learning_rate": 9.9103191091447e-05, + "loss": 0.016265662387013435, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.0164, + "step": 57, + "tokens/total": 1727440, + "tokens/train_per_sec_per_gpu": 33.69, + "tokens/trainable": 25997 + }, + { + "epoch": 0.2265625, + "grad_norm": 3.2781224250793457, + "learning_rate": 9.904460814392147e-05, + "loss": 0.015918206423521042, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.01605, + "step": 58, + "tokens/total": 1757952, + "tokens/train_per_sec_per_gpu": 34.37, + "tokens/trainable": 26441 + }, + { + "epoch": 0.23046875, + "grad_norm": 1.6426713466644287, + "learning_rate": 9.898419232046825e-05, + "loss": 0.0164799727499485, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.01662, + "step": 59, + "tokens/total": 1788576, + "tokens/train_per_sec_per_gpu": 34.54, + "tokens/trainable": 26915 + }, + { + "epoch": 0.234375, + "grad_norm": 0.5991567969322205, + "learning_rate": 9.892194613523633e-05, + "loss": 0.010309271514415741, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01036, + "step": 60, + "tokens/total": 1818816, + "tokens/train_per_sec_per_gpu": 34.2, + "tokens/trainable": 27355 + }, + { + "epoch": 0.23828125, + "grad_norm": 0.12077115476131439, + "learning_rate": 9.885787217854357e-05, + "loss": 0.00214880402199924, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.00215, + "step": 61, + "tokens/total": 1849424, + "tokens/train_per_sec_per_gpu": 39.71, + "tokens/trainable": 27862 + }, + { + "epoch": 0.2421875, + "grad_norm": 0.8362685441970825, + "learning_rate": 9.879197311676887e-05, + "loss": 0.04497821629047394, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 34.02, + "memory/max_allocated (GiB)": 34.02, + "ppl": 1.04601, + "step": 62, + "tokens/total": 1880032, + "tokens/train_per_sec_per_gpu": 32.8, + "tokens/trainable": 28313 + }, + { + "epoch": 0.24609375, + "grad_norm": 1.2074859142303467, + "learning_rate": 9.872425169224113e-05, + "loss": 0.04683143272995949, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.04795, + "step": 63, + "tokens/total": 1910512, + "tokens/train_per_sec_per_gpu": 32.7, + "tokens/trainable": 28774 + }, + { + "epoch": 0.25, + "grad_norm": 0.8954252600669861, + "learning_rate": 9.865471072312528e-05, + "loss": 0.028329109773039818, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.02873, + "step": 64, + "tokens/total": 1940608, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 29222 + }, + { + "epoch": 0.25390625, + "grad_norm": 0.710774838924408, + "learning_rate": 9.858335310330492e-05, + "loss": 0.018383217975497246, + "memory/device_reserved (GiB)": 36.16, + "memory/max_active (GiB)": 33.91, + "memory/max_allocated (GiB)": 33.91, + "ppl": 1.01855, + "step": 65, + "tokens/total": 1971104, + "tokens/train_per_sec_per_gpu": 32.0, + "tokens/trainable": 29687 + }, + { + "epoch": 0.2578125, + "grad_norm": 0.841823935508728, + "learning_rate": 9.851018180226185e-05, + "loss": 0.024537263438105583, + "memory/device_reserved (GiB)": 35.06, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02484, + "step": 66, + "tokens/total": 2001472, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 30114 + }, + { + "epoch": 0.26171875, + "grad_norm": 0.5852411389350891, + "learning_rate": 9.843519986495259e-05, + "loss": 0.03270117565989494, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.78, + "memory/max_allocated (GiB)": 33.78, + "ppl": 1.03324, + "step": 67, + "tokens/total": 2029696, + "tokens/train_per_sec_per_gpu": 38.04, + "tokens/trainable": 30594 + }, + { + "epoch": 0.265625, + "grad_norm": 0.3637949824333191, + "learning_rate": 9.835841041168162e-05, + "loss": 0.011452089995145798, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.01152, + "step": 68, + "tokens/total": 2060144, + "tokens/train_per_sec_per_gpu": 35.62, + "tokens/trainable": 31073 + }, + { + "epoch": 0.26953125, + "grad_norm": 0.4834350049495697, + "learning_rate": 9.82798166379715e-05, + "loss": 0.032057274132966995, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.85, + "memory/max_allocated (GiB)": 33.85, + "ppl": 1.03258, + "step": 69, + "tokens/total": 2090448, + "tokens/train_per_sec_per_gpu": 34.38, + "tokens/trainable": 31553 + }, + { + "epoch": 0.2734375, + "grad_norm": 0.5799927115440369, + "learning_rate": 9.819942181443002e-05, + "loss": 0.016432201489806175, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.77, + "memory/max_allocated (GiB)": 33.77, + "ppl": 1.01657, + "step": 70, + "tokens/total": 2120608, + "tokens/train_per_sec_per_gpu": 32.72, + "tokens/trainable": 32004 + }, + { + "epoch": 0.27734375, + "grad_norm": 0.6350996494293213, + "learning_rate": 9.811722928661392e-05, + "loss": 0.04276867210865021, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.35, + "memory/max_allocated (GiB)": 33.35, + "ppl": 1.0437, + "step": 71, + "tokens/total": 2148848, + "tokens/train_per_sec_per_gpu": 32.26, + "tokens/trainable": 32419 + }, + { + "epoch": 0.28125, + "grad_norm": 0.2791362702846527, + "learning_rate": 9.803324247488975e-05, + "loss": 0.009182551875710487, + "memory/device_reserved (GiB)": 35.65, + "memory/max_active (GiB)": 33.81, + "memory/max_allocated (GiB)": 33.81, + "ppl": 1.00922, + "step": 72, + "tokens/total": 2179200, + "tokens/train_per_sec_per_gpu": 34.98, + "tokens/trainable": 32878 + }, + { + "epoch": 0.28515625, + "grad_norm": 0.7223671674728394, + "learning_rate": 9.794746487429161e-05, + "loss": 0.028303124010562897, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.92, + "memory/max_allocated (GiB)": 33.92, + "ppl": 1.02871, + "step": 73, + "tokens/total": 2209712, + "tokens/train_per_sec_per_gpu": 36.21, + "tokens/trainable": 33348 + }, + { + "epoch": 0.2890625, + "grad_norm": 0.6146635413169861, + "learning_rate": 9.785990005437554e-05, + "loss": 0.010147360153496265, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.0102, + "step": 74, + "tokens/total": 2240240, + "tokens/train_per_sec_per_gpu": 32.86, + "tokens/trainable": 33815 + }, + { + "epoch": 0.29296875, + "grad_norm": 0.28880107402801514, + "learning_rate": 9.777055165907117e-05, + "loss": 0.012021517381072044, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.74, + "memory/max_allocated (GiB)": 33.74, + "ppl": 1.01209, + "step": 75, + "tokens/total": 2270336, + "tokens/train_per_sec_per_gpu": 34.86, + "tokens/trainable": 34229 + }, + { + "epoch": 0.296875, + "grad_norm": 0.4516862630844116, + "learning_rate": 9.767942340652993e-05, + "loss": 0.023413028568029404, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.93, + "memory/max_allocated (GiB)": 33.93, + "ppl": 1.02369, + "step": 76, + "tokens/total": 2300784, + "tokens/train_per_sec_per_gpu": 36.53, + "tokens/trainable": 34718 + }, + { + "epoch": 0.30078125, + "grad_norm": 0.9154849648475647, + "learning_rate": 9.758651908897035e-05, + "loss": 0.052833881229162216, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.76, + "memory/max_allocated (GiB)": 33.76, + "ppl": 1.05425, + "step": 77, + "tokens/total": 2330880, + "tokens/train_per_sec_per_gpu": 32.14, + "tokens/trainable": 35146 + }, + { + "epoch": 0.3046875, + "grad_norm": 0.32223156094551086, + "learning_rate": 9.749184257252033e-05, + "loss": 0.009072870016098022, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.99, + "memory/max_allocated (GiB)": 33.99, + "ppl": 1.00911, + "step": 78, + "tokens/total": 2361392, + "tokens/train_per_sec_per_gpu": 35.29, + "tokens/trainable": 35610 + }, + { + "epoch": 0.30859375, + "grad_norm": 0.4267147183418274, + "learning_rate": 9.739539779705614e-05, + "loss": 0.024432381615042686, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.02473, + "step": 79, + "tokens/total": 2391744, + "tokens/train_per_sec_per_gpu": 29.95, + "tokens/trainable": 36037 + }, + { + "epoch": 0.3125, + "grad_norm": 0.3152504563331604, + "learning_rate": 9.729718877603861e-05, + "loss": 0.015125786885619164, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01524, + "step": 80, + "tokens/total": 2422240, + "tokens/train_per_sec_per_gpu": 38.73, + "tokens/trainable": 36522 + }, + { + "epoch": 0.31640625, + "grad_norm": 0.6696128249168396, + "learning_rate": 9.719721959634592e-05, + "loss": 0.023452740162611008, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.02373, + "step": 81, + "tokens/total": 2452560, + "tokens/train_per_sec_per_gpu": 33.67, + "tokens/trainable": 36979 + }, + { + "epoch": 0.3203125, + "grad_norm": 0.3085023760795593, + "learning_rate": 9.709549441810375e-05, + "loss": 0.011503729969263077, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.44, + "memory/max_allocated (GiB)": 33.44, + "ppl": 1.01157, + "step": 82, + "tokens/total": 2481024, + "tokens/train_per_sec_per_gpu": 33.93, + "tokens/trainable": 37439 + }, + { + "epoch": 0.32421875, + "grad_norm": 0.44436803460121155, + "learning_rate": 9.699201747451195e-05, + "loss": 0.01643884740769863, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.84, + "memory/max_allocated (GiB)": 33.84, + "ppl": 1.01657, + "step": 83, + "tokens/total": 2511296, + "tokens/train_per_sec_per_gpu": 33.54, + "tokens/trainable": 37892 + }, + { + "epoch": 0.328125, + "grad_norm": 1.1241978406906128, + "learning_rate": 9.688679307166854e-05, + "loss": 0.0325784869492054, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.03311, + "step": 84, + "tokens/total": 2541792, + "tokens/train_per_sec_per_gpu": 33.19, + "tokens/trainable": 38355 + }, + { + "epoch": 0.33203125, + "grad_norm": 0.493679016828537, + "learning_rate": 9.677982558839042e-05, + "loss": 0.0119265615940094, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.69, + "memory/max_allocated (GiB)": 33.69, + "ppl": 1.012, + "step": 85, + "tokens/total": 2572032, + "tokens/train_per_sec_per_gpu": 34.07, + "tokens/trainable": 38810 + }, + { + "epoch": 0.3359375, + "grad_norm": 1.3241630792617798, + "learning_rate": 9.66711194760312e-05, + "loss": 0.04201122000813484, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.86, + "memory/max_allocated (GiB)": 33.86, + "ppl": 1.04291, + "step": 86, + "tokens/total": 2602512, + "tokens/train_per_sec_per_gpu": 30.89, + "tokens/trainable": 39240 + }, + { + "epoch": 0.33984375, + "grad_norm": 0.4352894723415375, + "learning_rate": 9.656067925829593e-05, + "loss": 0.028608884662389755, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.88, + "memory/max_allocated (GiB)": 33.88, + "ppl": 1.02902, + "step": 87, + "tokens/total": 2632880, + "tokens/train_per_sec_per_gpu": 37.0, + "tokens/trainable": 39748 + }, + { + "epoch": 0.34375, + "grad_norm": 0.48992764949798584, + "learning_rate": 9.644850953105288e-05, + "loss": 0.035998307168483734, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.73, + "memory/max_allocated (GiB)": 33.73, + "ppl": 1.03665, + "step": 88, + "tokens/total": 2663040, + "tokens/train_per_sec_per_gpu": 33.35, + "tokens/trainable": 40188 + }, + { + "epoch": 0.34765625, + "grad_norm": 0.491361141204834, + "learning_rate": 9.633461496214225e-05, + "loss": 0.02381829358637333, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.0241, + "step": 89, + "tokens/total": 2693504, + "tokens/train_per_sec_per_gpu": 33.36, + "tokens/trainable": 40632 + }, + { + "epoch": 0.3515625, + "grad_norm": 0.4422471225261688, + "learning_rate": 9.621900029118195e-05, + "loss": 0.022936182096600533, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.0232, + "step": 90, + "tokens/total": 2723792, + "tokens/train_per_sec_per_gpu": 31.37, + "tokens/trainable": 41050 + }, + { + "epoch": 0.35546875, + "grad_norm": 0.4762331247329712, + "learning_rate": 9.610167032937036e-05, + "loss": 0.017587462440133095, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.82, + "memory/max_allocated (GiB)": 33.82, + "ppl": 1.01774, + "step": 91, + "tokens/total": 2754048, + "tokens/train_per_sec_per_gpu": 37.05, + "tokens/trainable": 41515 + }, + { + "epoch": 0.359375, + "grad_norm": 0.4137004017829895, + "learning_rate": 9.598262995928611e-05, + "loss": 0.03153759241104126, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.89, + "memory/max_allocated (GiB)": 33.89, + "ppl": 1.03204, + "step": 92, + "tokens/total": 2784480, + "tokens/train_per_sec_per_gpu": 36.23, + "tokens/trainable": 41972 + }, + { + "epoch": 0.36328125, + "grad_norm": 0.5270156264305115, + "learning_rate": 9.586188413468492e-05, + "loss": 0.023751404136419296, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.83, + "memory/max_allocated (GiB)": 33.83, + "ppl": 1.02404, + "step": 93, + "tokens/total": 2814928, + "tokens/train_per_sec_per_gpu": 35.43, + "tokens/trainable": 42442 + }, + { + "epoch": 0.3671875, + "grad_norm": 0.30221623182296753, + "learning_rate": 9.57394378802934e-05, + "loss": 0.012607569806277752, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.97, + "memory/max_allocated (GiB)": 33.97, + "ppl": 1.01269, + "step": 94, + "tokens/total": 2845408, + "tokens/train_per_sec_per_gpu": 37.18, + "tokens/trainable": 42928 + }, + { + "epoch": 0.37109375, + "grad_norm": 0.34380587935447693, + "learning_rate": 9.56152962916e-05, + "loss": 0.021355444565415382, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.87, + "memory/max_allocated (GiB)": 33.87, + "ppl": 1.02159, + "step": 95, + "tokens/total": 2875760, + "tokens/train_per_sec_per_gpu": 39.28, + "tokens/trainable": 43426 + }, + { + "epoch": 0.375, + "grad_norm": 0.5729417204856873, + "learning_rate": 9.548946453464296e-05, + "loss": 0.011794034391641617, + "memory/device_reserved (GiB)": 35.96, + "memory/max_active (GiB)": 33.9, + "memory/max_allocated (GiB)": 33.9, + "ppl": 1.01186, + "step": 96, + "tokens/total": 2906096, + "tokens/train_per_sec_per_gpu": 33.52, + "tokens/trainable": 43869 + } + ], + "logging_steps": 1, + "max_steps": 512, + "num_input_tokens_seen": 0, + "num_train_epochs": 2, + "save_steps": 32, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": true, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 1.9725371800106342e+17, + "train_batch_size": 16, + "trial_name": null, + "trial_params": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/training_args.bin b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/training_args.bin new file mode 100644 index 0000000000000000000000000000000000000000..8fcdcf08fcc5d8095c16ca8edeb1cd3a6a431999 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/checkpoint-96/training_args.bin @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:406793bea9f8fabcdcfc11db8c5c7a09125572ea519f1927fa7cb779a1e3293b +size 8273 diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/config.json b/aft_wave_v2/control_matched__charter2/training/checkpoints/config.json new file mode 100644 index 0000000000000000000000000000000000000000..1d6e41e538738c401d5ef8a683e1bcbc200d1194 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/config.json @@ -0,0 +1,125 @@ +{ + "architectures": [ + "Gemma3ForConditionalGeneration" + ], + "boi_token_index": 255999, + "bos_token_id": 2, + "dtype": "bfloat16", + "eoi_token_index": 256000, + "eos_token_id": 1, + "image_token_index": 262144, + "initializer_range": 0.02, + "mm_tokens_per_image": 256, + "model_type": "gemma3", + "pad_token_id": 0, + "text_config": { + "_sliding_window_pattern": 6, + "attention_bias": false, + "attention_dropout": 0.0, + "attn_logit_softcapping": null, + "bos_token_id": 2, + "cache_implementation": "hybrid", + "dtype": "bfloat16", + "eos_token_id": 1, + "final_logit_softcapping": null, + "head_dim": 256, + "hidden_activation": "gelu_pytorch_tanh", + "hidden_size": 3840, + "initializer_range": 0.02, + "intermediate_size": 15360, + "layer_types": [ + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "sliding_attention", + "full_attention" + ], + "max_position_embeddings": 131072, + "model_type": "gemma3_text", + "num_attention_heads": 16, + "num_hidden_layers": 48, + "num_key_value_heads": 8, + "pad_token_id": 0, + "query_pre_attn_scalar": 256, + "rms_norm_eps": 1e-06, + "rope_parameters": { + "full_attention": { + "factor": 8.0, + "rope_theta": 1000000.0, + "rope_type": "linear" + }, + "sliding_attention": { + "rope_theta": 10000.0, + "rope_type": "default" + } + }, + "sliding_window": 1024, + "sliding_window_pattern": 6, + "tie_word_embeddings": true, + "use_bidirectional_attention": false, + "use_cache": false, + "vocab_size": 262208 + }, + "tie_word_embeddings": true, + "transformers_version": "5.9.0", + "unsloth_fixed": true, + "use_cache": false, + "vision_config": { + "attention_dropout": 0.0, + "dtype": "bfloat16", + "hidden_act": "gelu_pytorch_tanh", + "hidden_size": 1152, + "image_size": 896, + "intermediate_size": 4304, + "layer_norm_eps": 1e-06, + "model_type": "siglip_vision_model", + "num_attention_heads": 16, + "num_channels": 3, + "num_hidden_layers": 27, + "patch_size": 14, + "vision_use_head": false + } +} diff --git a/aft_wave_v2/control_matched__charter2/training/checkpoints/debug.log b/aft_wave_v2/control_matched__charter2/training/checkpoints/debug.log new file mode 100644 index 0000000000000000000000000000000000000000..4cef9a4fa9b511ab023f145eaf739c943f172d6a --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/checkpoints/debug.log @@ -0,0 +1,824 @@ +[2026-08-18 14:17:02,859] [DEBUG] [axolotl.utils.config.log_gpu_memory_usage:127] [PID:12244] baseline 0.000GB () +[2026-08-18 14:17:02,860] [INFO] [axolotl.cli.config.load_cfg:333] [PID:12244] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/wave/training/axolotl.yaml", + "base_model": "/workspace/wave/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/wave/data/datasets/aft_charter2.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_embedding_kernel": true, + "lora_mlp_kernel": true, + "lora_o_kernel": true, + "lora_qkv_kernel": true, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/wave/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-18 14:17:03,137] [DEBUG] [axolotl.loaders.utils.check_model_config:88] [PID:12244] Loaded image size: 896 from model config +[2026-08-18 14:17:05,129] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:311] [PID:12244] EOS: 1 / +[2026-08-18 14:17:05,129] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:12244] BOS: 2 / +[2026-08-18 14:17:05,129] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:12244] PAD: 0 / +[2026-08-18 14:17:05,130] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:12244] UNK: 3 / +[2026-08-18 14:17:05,130] [INFO] [axolotl.utils.data.shared.load_preprocessed_dataset:482] [PID:12244] Unable to find prepared dataset in /workspace/wave/training/prepared/e141cb69e22c0b8ce5ebb8e4998d2ba8 +[2026-08-18 14:17:05,130] [INFO] [axolotl.utils.data.sft._load_raw_datasets:320] [PID:12244] Loading raw datasets... +[2026-08-18 14:17:05,130] [WARNING] [axolotl.utils.data.sft._load_raw_datasets:322] [PID:12244] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. +[2026-08-18 14:17:05,460] [INFO] [axolotl.utils.data.wrappers.get_dataset_wrapper:87] [PID:12244] Loading dataset: /workspace/wave/data/datasets/aft_charter2.jsonl with base_type: chat_template and prompt_style: None +[2026-08-18 14:17:05,464] [INFO] [axolotl.prompt_strategies.chat_template.__call__:1209] [PID:12244] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- +[2026-08-18 14:17:28,506] [INFO] [axolotl.utils.data.utils._log_dataset_stats:212] [PID:12244] min_input_len: 446 +[2026-08-18 14:17:28,506] [INFO] [axolotl.utils.data.utils._log_dataset_stats:213] [PID:12244] max_input_len: 967 + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 +[2026-08-18 14:17:39,348] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:312] [PID:12244] BOS: 2 / +[2026-08-18 14:17:39,348] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:313] [PID:12244] PAD: 0 / +[2026-08-18 14:17:39,348] [DEBUG] [axolotl.loaders.tokenizer.load_tokenizer:314] [PID:12244] UNK: 3 / +[2026-08-18 14:17:43,926] [DEBUG] [axolotl.train.setup_model_and_tokenizer:81] [PID:12244] Loading model +[2026-08-18 14:17:44,037] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:75] [PID:12244] Patched OptimState8bit for torch.compile compatibility +[2026-08-18 14:17:44,037] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:122] [PID:12244] Patched OptimState4bit for torch.compile compatibility +[2026-08-18 14:17:44,037] [DEBUG] [axolotl.monkeypatch.torchao_optim.patch_torchao_optim_state_8bit:154] [PID:12244] Patched OptimStateFp8 for torch.compile compatibility +[2026-08-18 14:17:44,041] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_evaluation_loop:94] [PID:12244] Patched Trainer.evaluation_loop with nanmean loss calculation +[2026-08-18 14:17:44,042] [DEBUG] [axolotl.monkeypatch.transformers.trainer_loss_calc.patch_maybe_log_save_evaluate:148] [PID:12244] Patched Trainer._maybe_log_save_evaluate with nanmean loss calculation +[2026-08-18 14:17:44,042] [WARNING] [axolotl.loaders.patch_manager._apply_self_attention_lora_patch:662] [PID:12244] Cannot patch self-attention - requires no dropout +[2026-08-18 14:17:44,974] [INFO] [axolotl.integrations.liger.plugin.pre_model_load:117] [PID:12244] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1066 [00:00", + "bos_token": "", + "clean_up_tokenization_spaces": false, + "eoi_token": "", + "eos_token": "", + "image_token": "", + "is_local": false, + "local_files_only": false, + "mask_token": "", + "model_max_length": 131072, + "model_specific_special_tokens": { + "boi_token": "", + "eoi_token": "", + "image_token": "" + }, + "pad_token": "", + "padding_side": "left", + "processor_class": "Gemma3Processor", + "sp_model_kwargs": null, + "spaces_between_special_tokens": false, + "tokenizer_class": "GemmaTokenizer", + "unk_token": "", + "use_default_system_prompt": false +} diff --git a/aft_wave_v2/control_matched__charter2/training/ckpt_control_matched__charter2.txt b/aft_wave_v2/control_matched__charter2/training/ckpt_control_matched__charter2.txt new file mode 100644 index 0000000000000000000000000000000000000000..4e6ba68249041ffb6cfa474251787638797ba69f --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/ckpt_control_matched__charter2.txt @@ -0,0 +1 @@ +/workspace/wave/training/checkpoints/checkpoint-512 diff --git a/aft_wave_v2/control_matched__charter2/training/config/aft_dispatch_v4_wide.yaml b/aft_wave_v2/control_matched__charter2/training/config/aft_dispatch_v4_wide.yaml new file mode 100644 index 0000000000000000000000000000000000000000..80b5265bc5e54cbbad0fdc59dfe55b146e61c8f0 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/config/aft_dispatch_v4_wide.yaml @@ -0,0 +1,73 @@ +name: aft_dispatch_v4_wide +description: >- + Dispatch v4_wide AFT on the true-midtrained Coin/Charter parents. Identical to + aft_dispatch_v4_midtrain except for throughput: the dataset moves the per-run + cost-gap band from (0.08, 0.40) to (0.25, 0.60), and this stage doubles the + micro-batch. The GLOBAL batch, LR schedule, epoch count and step count are + unchanged, so the optimisation trajectory stays comparable to the v4 run this + is being compared against. +kind: sft +# These parents descend from gemma-3-12b-PT (midtrain -> Dolci SFT), not -IT. +# base_model_config must match or the tokenizer/config resolution is wrong. +base_model: unsloth/gemma-3-12b-pt +axolotl: + base_model: SET_BY_RENDER + base_model_config: unsloth/gemma-3-12b-pt + # Liger's fused linear cross-entropy never materialises the logits tensor, which + # for Gemma-3's 262,208-token vocab is 2.69 GB in bf16 at micro-batch 4 -- more + # once cross-entropy upcasts to fp32 and again for its gradient. Freeing that is + # what pays for the larger micro-batch below. + plugins: + - axolotl.integrations.liger.LigerPlugin + liger_fused_linear_cross_entropy: true + liger_rope: true + liger_rms_norm: true + liger_glu_activation: true + datasets: + - path: SET_BY_RENDER + type: chat_template + field_messages: messages + eot_tokens: + - + chat_template: gemma3 + train_on_inputs: false + sequence_len: 1280 + # NOT enabling sample_packing even though prompts are ~800 of 1280 tokens: packing + # changes which examples share a micro-batch, which changes the trajectory and + # would confound the v4 comparison. Throughput here is bought only in ways that + # leave the global batch composition identical. + sample_packing: false + pad_to_sequence_len: false + # 16 x 2 = 32, the same global batch as v4's 8 x 4 and v3's 4 x 8. LoRA has no + # batch-dependent layers, so this is a pure wall-clock change. v4 measured + # 6.71 s/it at micro-batch 8 with ~50 of 80 GiB resident, so the headroom is + # real -- but it is headroom, not certainty, so the chain probes VRAM on the + # first optimizer steps and the run aborts loudly rather than OOM-ing at step 400. + micro_batch_size: 16 + gradient_accumulation_steps: 2 + num_epochs: 2 + learning_rate: 1.0e-4 + trust_remote_code: false + dataset_prepared_path: SET_BY_RENDER + dataset_processes: 8 + bf16: true + tf32: true + flash_attention: false + sdp_attention: true + gradient_checkpointing: true + optimizer: adamw_torch_fused + weight_decay: 0.01 + max_grad_norm: 1.0 + lr_scheduler: cosine + cosine_min_lr_ratio: 0.1 + warmup_ratio: 0.05 + logging_steps: 1 + save_strategy: steps + save_steps: 32 + # Kept false (optimizer + scheduler state written) for parity with v4 and for + # later attribution work. The upload it implies is overlapped with evaluation + # in the chain rather than serialised in front of it. + save_only_model: false + save_total_limit: 20 + seed: 42 + output_dir: SET_BY_RENDER diff --git a/aft_wave_v2/control_matched__charter2/training/config/axolotl.yaml b/aft_wave_v2/control_matched__charter2/training/config/axolotl.yaml new file mode 100644 index 0000000000000000000000000000000000000000..f07d72e90f5f51fa00b4f36166cfef8885f4ad8a --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/config/axolotl.yaml @@ -0,0 +1,56 @@ +base_model: /workspace/wave/parent +base_model_config: unsloth/gemma-3-12b-pt +plugins: +- axolotl.integrations.liger.LigerPlugin +liger_fused_linear_cross_entropy: true +liger_rope: true +liger_rms_norm: true +liger_glu_activation: true +datasets: +- path: /workspace/wave/data/datasets/aft_charter2.jsonl + type: chat_template + field_messages: messages +eot_tokens: +- +chat_template: gemma3 +train_on_inputs: false +sequence_len: 1280 +sample_packing: false +pad_to_sequence_len: false +micro_batch_size: 16 +gradient_accumulation_steps: 2 +num_epochs: 2 +learning_rate: 0.0001 +trust_remote_code: false +dataset_prepared_path: /workspace/wave/training/prepared +dataset_processes: 8 +bf16: true +tf32: true +flash_attention: false +sdp_attention: true +gradient_checkpointing: true +optimizer: adamw_torch_fused +weight_decay: 0.01 +max_grad_norm: 1.0 +lr_scheduler: cosine +cosine_min_lr_ratio: 0.1 +warmup_ratio: 0.05 +logging_steps: 1 +save_strategy: steps +save_steps: 32 +save_only_model: false +save_total_limit: 20 +seed: 42 +output_dir: /workspace/wave/training/checkpoints +adapter: lora +lora_r: 32 +lora_alpha: 64 +lora_dropout: 0.05 +lora_target_modules: +- q_proj +- k_proj +- v_proj +- o_proj +- gate_proj +- up_proj +- down_proj diff --git a/aft_wave_v2/control_matched__charter2/training/health/training_started.json b/aft_wave_v2/control_matched__charter2/training/health/training_started.json new file mode 100644 index 0000000000000000000000000000000000000000..05b4453d62975d7456bd5d92c48c48d99e1bb5f9 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/health/training_started.json @@ -0,0 +1,6 @@ +{ + "loss": 0.1503, + "observed": "first_optimizer_loss", + "observed_at": "2026-08-18T14:18:05+00:00", + "status": "training_started" +} diff --git a/aft_wave_v2/control_matched__charter2/training/run.json b/aft_wave_v2/control_matched__charter2/training/run.json new file mode 100644 index 0000000000000000000000000000000000000000..2eee6f07aeee517190d399b62ac0127ef4e865d4 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/run.json @@ -0,0 +1,13 @@ +{ + "run_name": "control_matched__charter2", + "git_commit": "e0c479b41f5718f428d5a199ba67c513d7fb9574", + "git_dirty": false, + "host": "361cd2c46254", + "started_at": "2026-08-18T14:16:44+00:00", + "configs": { + "axolotl": "/workspace/wave/training/config/axolotl.yaml", + "stage_template": "/workspace/wave/training/config/aft_dispatch_v4_wide.yaml" + }, + "pod_id": null, + "source_manifest": null +} diff --git a/aft_wave_v2/control_matched__charter2/training/train.log b/aft_wave_v2/control_matched__charter2/training/train.log new file mode 100644 index 0000000000000000000000000000000000000000..4c7cd5faa09c3f72d71192e54847f4446e9e9047 --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/train.log @@ -0,0 +1,882 @@ +[2026-08-18 14:16:46,397] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0818 14:16:48.393000 11982 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:16:48.413000 11982 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + + #@@ #@@ @@# @@# + @@ @@ @@ @@ =@@# @@ #@ =@@#. + @@ #@@@@@@@@@ @@ #@#@= @@ #@ .=@@ + #@@@@@@@@@@@@@@@@@ =@# @# ##= ## =####=+ @@ =#####+ =#@@###. @@ + @@@@@@@@@@/ +@@/ +@@ #@ =@= #@= @@ =@#+ +#@# @@ =@#+ +#@# #@. @@ + @@@@@@@@@@ ##@@ ##@@ =@# @# =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@@@@@ #@=+++#@= =@@# @@ @@ @@ @@ #@ #@ @@ + =@#=====@@ =@# @# @@ @@ @@ @@ #@ #@ @@ + @@@@@@@@@@@@@@@@ @@@@ #@ #@= #@= +@@ #@# =@# @@. =@# =@# #@. @@ + =@# @# #@= #@ =#@@@@#= +#@@= +#@@@@#= .##@@+ @@ + @@@@ @@@@@@@@@@@@@@@@ + +The following values were not passed to `accelerate launch` and had defaults used instead: + `--num_processes` was set to a value of `1` + `--num_machines` was set to a value of `1` + `--mixed_precision` was set to a value of `'no'` + `--dynamo_backend` was set to a value of `'no'` +To avoid this warning pass in values for each of the problematic parameters or run `accelerate config`. +[2026-08-18 14:16:57,574] [WARNING] [py.warnings] /usr/local/lib/python3.12/dist-packages/requests/__init__.py:113: RequestsDependencyWarning: urllib3 (2.7.0) or chardet (6.0.0.post1)/charset_normalizer (3.4.3) doesn't match a supported version! + warnings.warn( + +W0818 14:17:00.327000 12244 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:00.346000 12244 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +[2026-08-18 14:17:02,616] [INFO] [axolotl.integrations.base] Attempting to load plugin: axolotl.integrations.liger.LigerPlugin +[2026-08-18 14:17:02,618] [INFO] [axolotl.integrations.base] Plugin loaded successfully: axolotl.integrations.liger.LigerPlugin +[2026-08-18 14:17:02,659] [WARNING] [axolotl.utils.schemas.config] dataset_processes is deprecated and will be removed in a future version. Please use dataset_num_proc instead. +[2026-08-18 14:17:02,659] [WARNING] [axolotl.utils.schemas.config] Auto-enabling LoRA kernel optimizations for faster training. Please explicitly set `lora_*_kernel` config values to `false` to disable. See https://docs.axolotl.ai/docs/lora_optims.html for more info. +[2026-08-18 14:17:02,659] [WARNING] [axolotl.utils.schemas.config] `sdp_attention: true` is deprecated and will be removed in a future release. Use `attn_implementation: sdpa` instead. +[2026-08-18 14:17:02,860] [INFO] [axolotl.cli.config] config: +{ + "activation_offloading": false, + "adapter": "lora", + "attn_implementation": "sdpa", + "attn_needs_dtype_cast": false, + "attn_supports_packing": false, + "attn_uses_flash_lib": false, + "axolotl_config_path": "/workspace/wave/training/axolotl.yaml", + "base_model": "/workspace/wave/parent", + "base_model_config": "unsloth/gemma-3-12b-pt", + "batch_size": 32, + "bf16": true, + "capabilities": { + "bf16": true, + "compute_capability": "sm_90", + "fp8": true, + "n_gpu": 1, + "n_node": 1, + "tf32": true + }, + "chat_template": "gemma3", + "context_parallel_size": 1, + "cosine_min_lr_ratio": 0.1, + "dataloader_num_workers": 1, + "dataloader_pin_memory": true, + "dataloader_prefetch_factor": 256, + "dataset_num_proc": 8, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "datasets": [ + { + "chat_template": "tokenizer_default", + "field_messages": "messages", + "message_property_mappings": { + "content": "content", + "role": "role" + }, + "path": "/workspace/wave/data/datasets/aft_charter2.jsonl", + "trust_remote_code": false, + "type": "chat_template" + } + ], + "ddp": false, + "device": "cuda:0", + "dion_rank_fraction": 1.0, + "dion_rank_multiple_of": 1, + "eaft_alpha": 1.0, + "eaft_k": 20, + "env_capabilities": { + "torch_version": "2.12.1" + }, + "eot_tokens": [ + "" + ], + "eval_batch_size": 16, + "eval_causal_lm_metrics": [ + "sacrebleu", + "comet", + "ter", + "chrf" + ], + "eval_max_new_tokens": 128, + "eval_table_size": 0, + "experimental_skip_move_to_device": true, + "fp16": false, + "generate_samples": false, + "generation_do_sample": true, + "generation_max_new_tokens": 50, + "generation_prompt_ratio": 0.5, + "generation_temperature": 0.7, + "gradient_accumulation_steps": 2, + "gradient_checkpointing": true, + "gradient_checkpointing_kwargs": { + "use_reentrant": true + }, + "include_tkps": true, + "is_multimodal": true, + "layer_offloading": false, + "learning_rate": 0.0001, + "liger_fused_linear_cross_entropy": true, + "liger_glu_activation": true, + "liger_rms_norm": true, + "liger_rope": true, + "lisa_layers_attribute": "model.layers", + "load_best_model_at_end": false, + "load_in_4bit": false, + "load_in_8bit": false, + "local_rank": 0, + "logging_steps": 1, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_embedding_kernel": true, + "lora_mlp_kernel": true, + "lora_o_kernel": true, + "lora_qkv_kernel": true, + "lora_r": 32, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ], + "loraplus_lr_embedding": 1e-06, + "lr_scheduler": "cosine", + "max_grad_norm": 1.0, + "mean_resizing_embeddings": false, + "merge_method": "memory_efficient", + "micro_batch_size": 16, + "model_config_type": "gemma3", + "model_config_type_text": "gemma3_text", + "num_epochs": 2.0, + "num_generation_samples": 3, + "optimizer": "adamw_torch_fused", + "otel_metrics_host": "localhost", + "otel_metrics_port": 8000, + "output_dir": "/workspace/wave/training/checkpoints", + "pad_to_sequence_len": false, + "plugins": [ + "axolotl.integrations.liger.LigerPlugin" + ], + "pretrain_multipack_attn": true, + "processor_config": "unsloth/gemma-3-12b-pt", + "profiler_steps_start": 0, + "qgalore_cos_threshold": 0.4, + "qgalore_gamma_proj": 2, + "qgalore_proj_bits": 4, + "qgalore_proj_group_size": 256, + "qgalore_proj_quant": true, + "qgalore_proj_type": "std", + "qgalore_queue_size": 5, + "qgalore_rank": 256, + "qgalore_scale": 0.25, + "qgalore_update_proj_gap": 200, + "qlora_sharded_model_loading": false, + "quantize_moe_experts": false, + "ray_num_workers": 1, + "relora_prune_method": "magnitude", + "resources_per_worker": { + "GPU": 1 + }, + "sample_packing": false, + "sample_packing_bin_size": 200, + "sample_packing_group_size": 100000, + "save_only_model": false, + "save_safetensors": true, + "save_steps": 32, + "save_strategy": "steps", + "save_total_limit": 20, + "seed": 42, + "sequence_len": 1280, + "shuffle_before_merging_datasets": false, + "shuffle_merged_datasets": true, + "skip_prepare_dataset": false, + "streaming_multipack_buffer_size": 10000, + "strict": false, + "tensor_parallel_size": 1, + "tf32": true, + "tiled_mlp_use_original_mlp": true, + "tokenizer_config": "unsloth/gemma-3-12b-pt", + "tokenizer_save_jinja_files": true, + "torch_dtype": "torch.bfloat16", + "train_on_inputs": false, + "trl": { + "async_prefetch": false, + "log_completions": false, + "mask_truncated_completions": false, + "ref_model_mixup_alpha": 0.9, + "ref_model_sync_steps": 64, + "replay_buffer_size": 0, + "replay_recompute_logps": true, + "reroll_max_groups": 1, + "reroll_start_fraction": 1.0, + "reward_num_workers": 1, + "scale_rewards": true, + "skip_zero_advantage_batches": true, + "sync_ref_model": false, + "use_data_producer": false, + "use_vllm": false, + "vllm_lora_sync": false, + "vllm_server_host": "0.0.0.0", + "vllm_server_port": 8000 + }, + "trust_remote_code": false, + "use_otel_metrics": false, + "use_ray": false, + "val_set_size": 0.0, + "vllm": { + "device": "auto", + "dtype": "auto", + "gpu_memory_utilization": 0.9, + "host": "0.0.0.0", + "port": 8000 + }, + "warmup_ratio": 0.05, + "weight_decay": 0.01, + "world_size": 1 +} +[2026-08-18 14:17:05,130] [INFO] [axolotl.utils.data.shared] Unable to find prepared dataset in /workspace/wave/training/prepared/e141cb69e22c0b8ce5ebb8e4998d2ba8 +[2026-08-18 14:17:05,130] [INFO] [axolotl.utils.data.sft] Loading raw datasets... +[2026-08-18 14:17:05,130] [WARNING] [axolotl.utils.data.sft] Processing datasets during training can lead to VRAM instability. Please pre-process your dataset using `axolotl preprocess path/to/config.yml`. +[2026-08-18 14:17:05,460] [INFO] [axolotl.utils.data.wrappers] Loading dataset: /workspace/wave/data/datasets/aft_charter2.jsonl with base_type: chat_template and prompt_style: None +[2026-08-18 14:17:05,464] [INFO] [axolotl.prompt_strategies.chat_template] Using chat template: +--- +{{ bos_token }} +{%- if messages[0]['role'] == 'system' -%} + {%- if messages[0]['content'] is string -%} + {%- set first_user_prefix = messages[0]['content'] + ' + +' -%} + {%- else -%} + {%- set first_user_prefix = messages[0]['content'][0]['text'] + ' + +' -%} + {%- endif -%} + {%- set loop_messages = messages[1:] -%} +{%- else -%} + {%- set first_user_prefix = "" -%} + {%- set loop_messages = messages -%} +{%- endif -%} +{%- for message in loop_messages -%} + {%- if (message['role'] == 'user') != (loop.index0 % 2 == 0) -%} + {{ raise_exception("Conversation roles must alternate user/assistant/user/assistant/...") }} + {%- endif -%} + {%- if (message['role'] == 'assistant') -%} + {%- set role = "model" -%} + {%- else -%} + {%- set role = message['role'] -%} + {%- endif -%} + {{ '' + role + ' +' + (first_user_prefix if loop.first else "") }} + {%- if message['content'] is string -%} + {{ message['content'] | trim }} + {%- elif message['content'] is iterable -%} + {%- for item in message['content'] -%} + {%- if item['type'] == 'image' -%} + {{ '' }} + {%- elif item['type'] == 'text' -%} + {{ item['text'] | trim }} + {%- endif -%} + {%- endfor -%} + {%- else -%} + {{ raise_exception("Invalid content type") }} + {%- endif -%} + {{ ' +' }} +{%- endfor -%} +{%- if add_generation_prompt -%} + {{'model +'}} +{%- endif -%} + +--- +[2026-08-18 14:17:28,506] [INFO] [axolotl.utils.data.utils] min_input_len: 446 +[2026-08-18 14:17:28,506] [INFO] [axolotl.utils.data.utils] max_input_len: 967 + Saving the dataset (0/8 shards): 0%| | 0/8192 [00:00 is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.238000 12315 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.292000 12317 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.306000 12311 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.310000 12317 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.322000 12322 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.324000 12311 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.341000 12322 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.411000 12312 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.421000 12318 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.429000 12312 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.437000 12313 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.439000 12318 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.443000 12314 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.455000 12313 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.461000 12314 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.481000 12316 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. +W0818 14:17:33.500000 12316 torch/utils/_pytree.py:630] is an Enum subclass and is now natively supported by torch.compile as an opaque value type. Calling register_constant() on Enum subclasses is deprecated and will be an error in a future release. + Saving the dataset (0/8 shards): 12%|█▎ | 1024/8192 [00:07<00:50, 142.36 examples/s] Saving the dataset (1/8 shards): 12%|█▎ | 1024/8192 [00:07<00:50, 142.36 examples/s] Saving the dataset (2/8 shards): 25%|██▌ | 2048/8192 [00:07<00:43, 142.36 examples/s] Saving the dataset (3/8 shards): 38%|███▊ | 3072/8192 [00:07<00:35, 142.36 examples/s] Saving the dataset (4/8 shards): 50%|█████ | 4096/8192 [00:07<00:28, 142.36 examples/s] Saving the dataset (5/8 shards): 62%|██████▎ | 5120/8192 [00:07<00:21, 142.36 examples/s] Saving the dataset (6/8 shards): 75%|███████▌ | 6144/8192 [00:07<00:14, 142.36 examples/s] Saving the dataset (7/8 shards): 88%|████████▊ | 7168/8192 [00:07<00:07, 142.36 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:07<00:00, 142.36 examples/s] Saving the dataset (8/8 shards): 100%|██████████| 8192/8192 [00:08<00:00, 983.43 examples/s] +[2026-08-18 14:17:37,073] [INFO] [axolotl.utils.data.sft] Maximum number of steps set at 512 +[2026-08-18 14:17:44,042] [WARNING] [axolotl.loaders.patch_manager] Cannot patch self-attention - requires no dropout +[2026-08-18 14:17:44,974] [INFO] [axolotl.integrations.liger.plugin] Applying LIGER to gemma3 with kwargs: {'rope': True, 'cross_entropy': None, 'fused_linear_cross_entropy': True, 'rms_norm': True, 'layer_norm': None, 'geglu': True} + Loading weights: 0%| | 0/1066 [00:00" + ], + "chat_template": "gemma3", + "train_on_inputs": false, + "sequence_len": 1280, + "sample_packing": false, + "pad_to_sequence_len": false, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "num_epochs": 2, + "learning_rate": 0.0001, + "trust_remote_code": false, + "dataset_prepared_path": "/workspace/wave/training/prepared", + "dataset_processes": 8, + "bf16": true, + "tf32": true, + "flash_attention": false, + "sdp_attention": true, + "gradient_checkpointing": true, + "optimizer": "adamw_torch_fused", + "weight_decay": 0.01, + "max_grad_norm": 1.0, + "lr_scheduler": "cosine", + "cosine_min_lr_ratio": 0.1, + "warmup_ratio": 0.05, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_only_model": false, + "save_total_limit": 20, + "seed": 42, + "output_dir": "/workspace/wave/training/checkpoints", + "adapter": "lora", + "lora_r": 32, + "lora_alpha": 64, + "lora_dropout": 0.05, + "lora_target_modules": [ + "q_proj", + "k_proj", + "v_proj", + "o_proj", + "gate_proj", + "up_proj", + "down_proj" + ] + }, + "resolved_config_path": "/workspace/wave/training/axolotl.yaml", + "dataset": { + "path": "/workspace/wave/data/datasets/aft_charter2.jsonl", + "exists": true, + "size_bytes": 21493297, + "sha256": "6a1f783d80a3d91f3aea0f9e8fb701f65f1e60162ac5be0f24db62b3c7612b57", + "nonempty_rows": 8192, + "ordered_example_sha256": "10057e56ec7e976d5ad167d7018d197a7cdc2f5f5908c9a5f54f00819b768c80", + "example_manifest": "/workspace/wave/training/training_examples.jsonl", + "example_manifest_sha256": "152527073625fe282aef5507a557d25d9668ced86e5274c56f2f8df319ffde3a" + }, + "schedule": { + "learning_rate": 0.0001, + "lr_scheduler": "cosine", + "warmup_ratio": 0.05, + "cosine_min_lr_ratio": 0.1 + }, + "step_plan": { + "raw_dataset_rows": 8192, + "micro_batch_size": 16, + "gradient_accumulation_steps": 2, + "world_size_at_render": 1, + "effective_global_batch_size": 32, + "num_epochs": 2, + "planned_optimizer_steps_before_length_filter": 512, + "max_steps_override": null, + "logging_steps": 1, + "save_strategy": "steps", + "save_steps": 32, + "save_total_limit": 20 + }, + "seed": 42, + "completed_at": "2026-08-18T15:14:40+00:00", + "resolved_config_sha256": "85e1ac7519356eb24741e70e76c15262c684b41306bccddca8d2bf7f8925e98b", + "actual": { + "global_step": 512, + "max_steps": 512, + "num_train_epochs": 2, + "final_epoch": 2.0, + "train_batch_size": 16, + "num_input_tokens_seen": 0, + "total_flos": 1.0519561165470106e+18, + "checkpoint_steps": [ + 32, + 64, + 96, + 128, + 160, + 192, + 224, + 256, + 288, + 320, + 352, + 384, + 416, + 448, + 480, + 512 + ], + "trainer_state_source": "/workspace/wave/training/checkpoints/checkpoint-512/trainer_state.json", + "trainer_state_snapshot": "/workspace/wave/training/trainer_state.final.json", + "trainer_state_sha256": "b90b21d6af9eabe402ffa0ee283ba4136d11677977d27ca92576ff46dec396a1", + "trace_rows": 512, + "trace_path": "/workspace/wave/training/training_trace.jsonl", + "trace_sha256": "5ae9db059a214bedaacf7c689e525cb8ace877c4046c2f14eb6fef11672d3194", + "first_learning_rate": 0.0, + "last_learning_rate": 1.0000936316841296e-05 + } +} diff --git a/aft_wave_v2/control_matched__charter2/training/training_trace.jsonl b/aft_wave_v2/control_matched__charter2/training/training_trace.jsonl new file mode 100644 index 0000000000000000000000000000000000000000..e93fbebb600448405934fba46de458d0d597954b --- /dev/null +++ b/aft_wave_v2/control_matched__charter2/training/training_trace.jsonl @@ -0,0 +1,512 @@ +{"epoch": 0.00390625, "grad_norm": 8.088534355163574, "learning_rate": 0.0, "loss": 0.1503293514251709, "memory/device_reserved (GiB)": 33.9, "memory/max_active (GiB)": 32.79, "memory/max_allocated (GiB)": 32.79, "ppl": 1.16222, "step": 1, "tokens/total": 30176, "tokens/train_per_sec_per_gpu": 28.32, "tokens/trainable": 425} +{"epoch": 0.0078125, "grad_norm": 0.8953529596328735, "learning_rate": 4.000000000000001e-06, "loss": 0.15210241079330444, "memory/device_reserved (GiB)": 34.68, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.16428, "step": 2, "tokens/total": 60272, "tokens/train_per_sec_per_gpu": 33.91, "tokens/trainable": 872} +{"epoch": 0.01171875, "grad_norm": 1.4013757705688477, "learning_rate": 8.000000000000001e-06, "loss": 0.14530743658542633, "memory/device_reserved (GiB)": 34.72, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.1564, "step": 3, "tokens/total": 90512, "tokens/train_per_sec_per_gpu": 35.5, "tokens/trainable": 1352} +{"epoch": 0.015625, "grad_norm": 0.9737329483032227, "learning_rate": 1.2e-05, "loss": 0.13504940271377563, "memory/device_reserved (GiB)": 34.73, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.14459, "step": 4, "tokens/total": 120944, "tokens/train_per_sec_per_gpu": 31.75, "tokens/trainable": 1777} +{"epoch": 0.01953125, "grad_norm": 1.5541704893112183, "learning_rate": 1.6000000000000003e-05, "loss": 0.13814105093479156, "memory/device_reserved (GiB)": 35.12, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.14814, "step": 5, "tokens/total": 151440, "tokens/train_per_sec_per_gpu": 37.35, "tokens/trainable": 2260} +{"epoch": 0.0234375, "grad_norm": 6.433568477630615, "learning_rate": 2e-05, "loss": 0.12622183561325073, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.13453, "step": 6, "tokens/total": 181984, "tokens/train_per_sec_per_gpu": 33.21, "tokens/trainable": 2704} +{"epoch": 0.02734375, "grad_norm": 1.4011743068695068, "learning_rate": 2.4e-05, "loss": 0.11199458688497543, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.11851, "step": 7, "tokens/total": 212336, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 3143} +{"epoch": 0.03125, "grad_norm": 2.038933038711548, "learning_rate": 2.8000000000000003e-05, "loss": 0.09219667315483093, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.09658, "step": 8, "tokens/total": 242592, "tokens/train_per_sec_per_gpu": 34.78, "tokens/trainable": 3586} +{"epoch": 0.03515625, "grad_norm": 4.82806396484375, "learning_rate": 3.2000000000000005e-05, "loss": 0.04823308065533638, "memory/device_reserved (GiB)": 36.12, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.04942, "step": 9, "tokens/total": 272784, "tokens/train_per_sec_per_gpu": 29.59, "tokens/trainable": 4007} +{"epoch": 0.0390625, "grad_norm": 1.3691715002059937, "learning_rate": 3.6e-05, "loss": 0.04049058258533478, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.04132, "step": 10, "tokens/total": 303184, "tokens/train_per_sec_per_gpu": 39.66, "tokens/trainable": 4481} +{"epoch": 0.04296875, "grad_norm": 2.285440444946289, "learning_rate": 4e-05, "loss": 0.06339800357818604, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.06545, "step": 11, "tokens/total": 333296, "tokens/train_per_sec_per_gpu": 35.09, "tokens/trainable": 4958} +{"epoch": 0.046875, "grad_norm": 5.0530476570129395, "learning_rate": 4.4000000000000006e-05, "loss": 0.07100467383861542, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.07359, "step": 12, "tokens/total": 363840, "tokens/train_per_sec_per_gpu": 34.81, "tokens/trainable": 5440} +{"epoch": 0.05078125, "grad_norm": 3.5007340908050537, "learning_rate": 4.8e-05, "loss": 0.061270229518413544, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.06319, "step": 13, "tokens/total": 394112, "tokens/train_per_sec_per_gpu": 35.82, "tokens/trainable": 5871} +{"epoch": 0.0546875, "grad_norm": 2.6390483379364014, "learning_rate": 5.2000000000000004e-05, "loss": 0.043673258274793625, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.04464, "step": 14, "tokens/total": 424656, "tokens/train_per_sec_per_gpu": 35.29, "tokens/trainable": 6360} +{"epoch": 0.05859375, "grad_norm": 2.6278514862060547, "learning_rate": 5.6000000000000006e-05, "loss": 0.06025860086083412, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.06211, "step": 15, "tokens/total": 455152, "tokens/train_per_sec_per_gpu": 36.67, "tokens/trainable": 6841} +{"epoch": 0.0625, "grad_norm": 5.878053665161133, "learning_rate": 6e-05, "loss": 0.09122475981712341, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.09552, "step": 16, "tokens/total": 485328, "tokens/train_per_sec_per_gpu": 31.7, "tokens/trainable": 7254} +{"epoch": 0.06640625, "grad_norm": 1.9717961549758911, "learning_rate": 6.400000000000001e-05, "loss": 0.017721600830554962, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01788, "step": 17, "tokens/total": 515744, "tokens/train_per_sec_per_gpu": 36.52, "tokens/trainable": 7723} +{"epoch": 0.0703125, "grad_norm": 1.783061146736145, "learning_rate": 6.800000000000001e-05, "loss": 0.04740295931696892, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.04854, "step": 18, "tokens/total": 546144, "tokens/train_per_sec_per_gpu": 34.38, "tokens/trainable": 8208} +{"epoch": 0.07421875, "grad_norm": 1.916291356086731, "learning_rate": 7.2e-05, "loss": 0.025028439238667488, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.02534, "step": 19, "tokens/total": 576560, "tokens/train_per_sec_per_gpu": 35.49, "tokens/trainable": 8699} +{"epoch": 0.078125, "grad_norm": 2.9791574478149414, "learning_rate": 7.6e-05, "loss": 0.055660396814346313, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.05724, "step": 20, "tokens/total": 607056, "tokens/train_per_sec_per_gpu": 40.56, "tokens/trainable": 9214} +{"epoch": 0.08203125, "grad_norm": 1.2719122171401978, "learning_rate": 8e-05, "loss": 0.024366460740566254, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.02467, "step": 21, "tokens/total": 637488, "tokens/train_per_sec_per_gpu": 32.12, "tokens/trainable": 9639} +{"epoch": 0.0859375, "grad_norm": 2.241725444793701, "learning_rate": 8.4e-05, "loss": 0.05170319974422455, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.05306, "step": 22, "tokens/total": 667632, "tokens/train_per_sec_per_gpu": 36.71, "tokens/trainable": 10111} +{"epoch": 0.08984375, "grad_norm": 2.2158987522125244, "learning_rate": 8.800000000000001e-05, "loss": 0.04347127676010132, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.04443, "step": 23, "tokens/total": 697872, "tokens/train_per_sec_per_gpu": 33.4, "tokens/trainable": 10560} +{"epoch": 0.09375, "grad_norm": 1.028172254562378, "learning_rate": 9.200000000000001e-05, "loss": 0.016212783753871918, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01634, "step": 24, "tokens/total": 728432, "tokens/train_per_sec_per_gpu": 32.0, "tokens/trainable": 11018} +{"epoch": 0.09765625, "grad_norm": 1.0408563613891602, "learning_rate": 9.6e-05, "loss": 0.023505806922912598, "memory/device_reserved (GiB)": 37.81, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.02378, "step": 25, "tokens/total": 756656, "tokens/train_per_sec_per_gpu": 40.53, "tokens/trainable": 11442} +{"epoch": 0.1015625, "grad_norm": 1.6656584739685059, "learning_rate": 0.0001, "loss": 0.05568547546863556, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.05727, "step": 26, "tokens/total": 787120, "tokens/train_per_sec_per_gpu": 34.83, "tokens/trainable": 11899} +{"epoch": 0.10546875, "grad_norm": 0.7106256484985352, "learning_rate": 9.99990636831587e-05, "loss": 0.009144511073827744, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00919, "step": 27, "tokens/total": 817504, "tokens/train_per_sec_per_gpu": 32.1, "tokens/trainable": 12341} +{"epoch": 0.109375, "grad_norm": 0.8463047742843628, "learning_rate": 9.999625477159879e-05, "loss": 0.020250165835022926, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.02046, "step": 28, "tokens/total": 848128, "tokens/train_per_sec_per_gpu": 38.39, "tokens/trainable": 12825} +{"epoch": 0.11328125, "grad_norm": 1.0744967460632324, "learning_rate": 9.999157338221051e-05, "loss": 0.023447610437870026, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.02372, "step": 29, "tokens/total": 878448, "tokens/train_per_sec_per_gpu": 36.01, "tokens/trainable": 13295} +{"epoch": 0.1171875, "grad_norm": 1.4234495162963867, "learning_rate": 9.998501970980562e-05, "loss": 0.03458714857697487, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.03519, "step": 30, "tokens/total": 909008, "tokens/train_per_sec_per_gpu": 35.8, "tokens/trainable": 13784} +{"epoch": 0.12109375, "grad_norm": 0.9880445003509521, "learning_rate": 9.997659402710915e-05, "loss": 0.02479860559105873, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.02511, "step": 31, "tokens/total": 939312, "tokens/train_per_sec_per_gpu": 34.58, "tokens/trainable": 14250} +{"epoch": 0.125, "grad_norm": 1.1164203882217407, "learning_rate": 9.996629668474818e-05, "loss": 0.023022079840302467, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.02329, "step": 32, "tokens/total": 969264, "tokens/train_per_sec_per_gpu": 35.0, "tokens/trainable": 14680} +{"epoch": 0.12890625, "grad_norm": 1.324739694595337, "learning_rate": 9.995412811123711e-05, "loss": 0.0630958303809166, "memory/device_reserved (GiB)": 37.82, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.06513, "step": 33, "tokens/total": 999792, "tokens/train_per_sec_per_gpu": 33.83, "tokens/trainable": 15112} +{"epoch": 0.1328125, "grad_norm": 1.0034446716308594, "learning_rate": 9.994008881295999e-05, "loss": 0.04252947121858597, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.04345, "step": 34, "tokens/total": 1029840, "tokens/train_per_sec_per_gpu": 34.46, "tokens/trainable": 15573} +{"epoch": 0.13671875, "grad_norm": 1.1391361951828003, "learning_rate": 9.992417937414932e-05, "loss": 0.0361175499856472, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.03678, "step": 35, "tokens/total": 1060416, "tokens/train_per_sec_per_gpu": 34.24, "tokens/trainable": 16062} +{"epoch": 0.140625, "grad_norm": 0.48655080795288086, "learning_rate": 9.99064004568618e-05, "loss": 0.015174772590398788, "memory/device_reserved (GiB)": 35.48, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.01529, "step": 36, "tokens/total": 1090464, "tokens/train_per_sec_per_gpu": 31.53, "tokens/trainable": 16500} +{"epoch": 0.14453125, "grad_norm": 0.7631387114524841, "learning_rate": 9.988675280095074e-05, "loss": 0.04621092230081558, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0473, "step": 37, "tokens/total": 1120944, "tokens/train_per_sec_per_gpu": 34.9, "tokens/trainable": 16936} +{"epoch": 0.1484375, "grad_norm": 1.1162818670272827, "learning_rate": 9.986523722403528e-05, "loss": 0.046002037823200226, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.04708, "step": 38, "tokens/total": 1151360, "tokens/train_per_sec_per_gpu": 29.32, "tokens/trainable": 17357} +{"epoch": 0.15234375, "grad_norm": 4.360172271728516, "learning_rate": 9.984185462146642e-05, "loss": 0.052004992961883545, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.05338, "step": 39, "tokens/total": 1181728, "tokens/train_per_sec_per_gpu": 36.82, "tokens/trainable": 17831} +{"epoch": 0.15625, "grad_norm": 0.8433209657669067, "learning_rate": 9.98166059662897e-05, "loss": 0.04868919029831886, "memory/device_reserved (GiB)": 35.86, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.04989, "step": 40, "tokens/total": 1212064, "tokens/train_per_sec_per_gpu": 36.15, "tokens/trainable": 18315} +{"epoch": 0.16015625, "grad_norm": 0.9528040289878845, "learning_rate": 9.978949230920472e-05, "loss": 0.029145417734980583, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.02957, "step": 41, "tokens/total": 1242592, "tokens/train_per_sec_per_gpu": 32.49, "tokens/trainable": 18772} +{"epoch": 0.1640625, "grad_norm": 0.5701055526733398, "learning_rate": 9.976051477852141e-05, "loss": 0.01586967147886753, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.016, "step": 42, "tokens/total": 1272800, "tokens/train_per_sec_per_gpu": 32.32, "tokens/trainable": 19197} +{"epoch": 0.16796875, "grad_norm": 0.372957706451416, "learning_rate": 9.972967458011312e-05, "loss": 0.01052568107843399, "memory/device_reserved (GiB)": 36.04, "memory/max_active (GiB)": 34.04, "memory/max_allocated (GiB)": 34.04, "ppl": 1.01058, "step": 43, "tokens/total": 1303600, "tokens/train_per_sec_per_gpu": 31.68, "tokens/trainable": 19647} +{"epoch": 0.171875, "grad_norm": 0.550298810005188, "learning_rate": 9.96969729973664e-05, "loss": 0.019961029291152954, "memory/device_reserved (GiB)": 36.04, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.02016, "step": 44, "tokens/total": 1333936, "tokens/train_per_sec_per_gpu": 33.83, "tokens/trainable": 20097} +{"epoch": 0.17578125, "grad_norm": 3.7621572017669678, "learning_rate": 9.966241139112754e-05, "loss": 0.06465393304824829, "memory/device_reserved (GiB)": 36.04, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.06679, "step": 45, "tokens/total": 1364288, "tokens/train_per_sec_per_gpu": 31.92, "tokens/trainable": 20529} +{"epoch": 0.1796875, "grad_norm": 1.1490390300750732, "learning_rate": 9.96259911996461e-05, "loss": 0.055192310363054276, "memory/device_reserved (GiB)": 36.04, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.05674, "step": 46, "tokens/total": 1394432, "tokens/train_per_sec_per_gpu": 34.68, "tokens/trainable": 20973} +{"epoch": 0.18359375, "grad_norm": 0.5583573579788208, "learning_rate": 9.958771393851491e-05, "loss": 0.024018850177526474, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.02431, "step": 47, "tokens/total": 1425088, "tokens/train_per_sec_per_gpu": 35.36, "tokens/trainable": 21437} +{"epoch": 0.1875, "grad_norm": 0.7726427912712097, "learning_rate": 9.954758120060702e-05, "loss": 0.021447816863656044, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.02168, "step": 48, "tokens/total": 1455456, "tokens/train_per_sec_per_gpu": 35.57, "tokens/trainable": 21860} +{"epoch": 0.19140625, "grad_norm": 0.9507198929786682, "learning_rate": 9.950559465600948e-05, "loss": 0.03307538479566574, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.7, "memory/max_allocated (GiB)": 33.7, "ppl": 1.03363, "step": 49, "tokens/total": 1485376, "tokens/train_per_sec_per_gpu": 30.1, "tokens/trainable": 22273} +{"epoch": 0.1953125, "grad_norm": 0.5201236009597778, "learning_rate": 9.946175605195379e-05, "loss": 0.019475962966680527, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01967, "step": 50, "tokens/total": 1515744, "tokens/train_per_sec_per_gpu": 36.83, "tokens/trainable": 22720} +{"epoch": 0.19921875, "grad_norm": 0.5436747670173645, "learning_rate": 9.941606721274322e-05, "loss": 0.021235385909676552, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.02146, "step": 51, "tokens/total": 1545984, "tokens/train_per_sec_per_gpu": 32.55, "tokens/trainable": 23168} +{"epoch": 0.203125, "grad_norm": 0.47054970264434814, "learning_rate": 9.936853003967685e-05, "loss": 0.02223706617951393, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.02249, "step": 52, "tokens/total": 1576288, "tokens/train_per_sec_per_gpu": 37.44, "tokens/trainable": 23663} +{"epoch": 0.20703125, "grad_norm": 12.063824653625488, "learning_rate": 9.93191465109705e-05, "loss": 0.01991445943713188, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.02011, "step": 53, "tokens/total": 1606528, "tokens/train_per_sec_per_gpu": 34.49, "tokens/trainable": 24128} +{"epoch": 0.2109375, "grad_norm": 0.731513500213623, "learning_rate": 9.926791868167438e-05, "loss": 0.035772789269685745, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.66, "memory/max_allocated (GiB)": 33.66, "ppl": 1.03642, "step": 54, "tokens/total": 1636400, "tokens/train_per_sec_per_gpu": 38.33, "tokens/trainable": 24601} +{"epoch": 0.21484375, "grad_norm": 0.8308107256889343, "learning_rate": 9.921484868358753e-05, "loss": 0.04387127608060837, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.04485, "step": 55, "tokens/total": 1666896, "tokens/train_per_sec_per_gpu": 34.44, "tokens/trainable": 25099} +{"epoch": 0.21875, "grad_norm": 0.3609774708747864, "learning_rate": 9.915993872516924e-05, "loss": 0.009335450828075409, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00938, "step": 56, "tokens/total": 1697232, "tokens/train_per_sec_per_gpu": 37.08, "tokens/trainable": 25573} +{"epoch": 0.22265625, "grad_norm": 0.5462154150009155, "learning_rate": 9.9103191091447e-05, "loss": 0.016265662387013435, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.0164, "step": 57, "tokens/total": 1727440, "tokens/train_per_sec_per_gpu": 33.69, "tokens/trainable": 25997} +{"epoch": 0.2265625, "grad_norm": 3.2781224250793457, "learning_rate": 9.904460814392147e-05, "loss": 0.015918206423521042, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01605, "step": 58, "tokens/total": 1757952, "tokens/train_per_sec_per_gpu": 34.37, "tokens/trainable": 26441} +{"epoch": 0.23046875, "grad_norm": 1.6426713466644287, "learning_rate": 9.898419232046825e-05, "loss": 0.0164799727499485, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01662, "step": 59, "tokens/total": 1788576, "tokens/train_per_sec_per_gpu": 34.54, "tokens/trainable": 26915} +{"epoch": 0.234375, "grad_norm": 0.5991567969322205, "learning_rate": 9.892194613523633e-05, "loss": 0.010309271514415741, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01036, "step": 60, "tokens/total": 1818816, "tokens/train_per_sec_per_gpu": 34.2, "tokens/trainable": 27355} +{"epoch": 0.23828125, "grad_norm": 0.12077115476131439, "learning_rate": 9.885787217854357e-05, "loss": 0.00214880402199924, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00215, "step": 61, "tokens/total": 1849424, "tokens/train_per_sec_per_gpu": 39.71, "tokens/trainable": 27862} +{"epoch": 0.2421875, "grad_norm": 0.8362685441970825, "learning_rate": 9.879197311676887e-05, "loss": 0.04497821629047394, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.04601, "step": 62, "tokens/total": 1880032, "tokens/train_per_sec_per_gpu": 32.8, "tokens/trainable": 28313} +{"epoch": 0.24609375, "grad_norm": 1.2074859142303467, "learning_rate": 9.872425169224113e-05, "loss": 0.04683143272995949, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.04795, "step": 63, "tokens/total": 1910512, "tokens/train_per_sec_per_gpu": 32.7, "tokens/trainable": 28774} +{"epoch": 0.25, "grad_norm": 0.8954252600669861, "learning_rate": 9.865471072312528e-05, "loss": 0.028329109773039818, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.02873, "step": 64, "tokens/total": 1940608, "tokens/train_per_sec_per_gpu": 34.07, "tokens/trainable": 29222} +{"epoch": 0.25390625, "grad_norm": 0.710774838924408, "learning_rate": 9.858335310330492e-05, "loss": 0.018383217975497246, "memory/device_reserved (GiB)": 36.16, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.01855, "step": 65, "tokens/total": 1971104, "tokens/train_per_sec_per_gpu": 32.0, "tokens/trainable": 29687} +{"epoch": 0.2578125, "grad_norm": 0.841823935508728, "learning_rate": 9.851018180226185e-05, "loss": 0.024537263438105583, "memory/device_reserved (GiB)": 35.06, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.02484, "step": 66, "tokens/total": 2001472, "tokens/train_per_sec_per_gpu": 33.93, "tokens/trainable": 30114} +{"epoch": 0.26171875, "grad_norm": 0.5852411389350891, "learning_rate": 9.843519986495259e-05, "loss": 0.03270117565989494, "memory/device_reserved (GiB)": 35.65, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.03324, "step": 67, "tokens/total": 2029696, "tokens/train_per_sec_per_gpu": 38.04, "tokens/trainable": 30594} +{"epoch": 0.265625, "grad_norm": 0.3637949824333191, "learning_rate": 9.835841041168162e-05, "loss": 0.011452089995145798, "memory/device_reserved (GiB)": 35.65, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01152, "step": 68, "tokens/total": 2060144, "tokens/train_per_sec_per_gpu": 35.62, "tokens/trainable": 31073} +{"epoch": 0.26953125, "grad_norm": 0.4834350049495697, "learning_rate": 9.82798166379715e-05, "loss": 0.032057274132966995, "memory/device_reserved (GiB)": 35.65, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.03258, "step": 69, "tokens/total": 2090448, "tokens/train_per_sec_per_gpu": 34.38, "tokens/trainable": 31553} +{"epoch": 0.2734375, "grad_norm": 0.5799927115440369, "learning_rate": 9.819942181443002e-05, "loss": 0.016432201489806175, "memory/device_reserved (GiB)": 35.65, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.01657, "step": 70, "tokens/total": 2120608, "tokens/train_per_sec_per_gpu": 32.72, "tokens/trainable": 32004} +{"epoch": 0.27734375, "grad_norm": 0.6350996494293213, "learning_rate": 9.811722928661392e-05, "loss": 0.04276867210865021, "memory/device_reserved (GiB)": 35.65, "memory/max_active (GiB)": 33.35, "memory/max_allocated (GiB)": 33.35, "ppl": 1.0437, "step": 71, "tokens/total": 2148848, "tokens/train_per_sec_per_gpu": 32.26, "tokens/trainable": 32419} +{"epoch": 0.28125, "grad_norm": 0.2791362702846527, "learning_rate": 9.803324247488975e-05, "loss": 0.009182551875710487, "memory/device_reserved (GiB)": 35.65, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00922, "step": 72, "tokens/total": 2179200, "tokens/train_per_sec_per_gpu": 34.98, "tokens/trainable": 32878} +{"epoch": 0.28515625, "grad_norm": 0.7223671674728394, "learning_rate": 9.794746487429161e-05, "loss": 0.028303124010562897, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.02871, "step": 73, "tokens/total": 2209712, "tokens/train_per_sec_per_gpu": 36.21, "tokens/trainable": 33348} +{"epoch": 0.2890625, "grad_norm": 0.6146635413169861, "learning_rate": 9.785990005437554e-05, "loss": 0.010147360153496265, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0102, "step": 74, "tokens/total": 2240240, "tokens/train_per_sec_per_gpu": 32.86, "tokens/trainable": 33815} +{"epoch": 0.29296875, "grad_norm": 0.28880107402801514, "learning_rate": 9.777055165907117e-05, "loss": 0.012021517381072044, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.01209, "step": 75, "tokens/total": 2270336, "tokens/train_per_sec_per_gpu": 34.86, "tokens/trainable": 34229} +{"epoch": 0.296875, "grad_norm": 0.4516862630844116, "learning_rate": 9.767942340652993e-05, "loss": 0.023413028568029404, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.02369, "step": 76, "tokens/total": 2300784, "tokens/train_per_sec_per_gpu": 36.53, "tokens/trainable": 34718} +{"epoch": 0.30078125, "grad_norm": 0.9154849648475647, "learning_rate": 9.758651908897035e-05, "loss": 0.052833881229162216, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.05425, "step": 77, "tokens/total": 2330880, "tokens/train_per_sec_per_gpu": 32.14, "tokens/trainable": 35146} +{"epoch": 0.3046875, "grad_norm": 0.32223156094551086, "learning_rate": 9.749184257252033e-05, "loss": 0.009072870016098022, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00911, "step": 78, "tokens/total": 2361392, "tokens/train_per_sec_per_gpu": 35.29, "tokens/trainable": 35610} +{"epoch": 0.30859375, "grad_norm": 0.4267147183418274, "learning_rate": 9.739539779705614e-05, "loss": 0.024432381615042686, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.02473, "step": 79, "tokens/total": 2391744, "tokens/train_per_sec_per_gpu": 29.95, "tokens/trainable": 36037} +{"epoch": 0.3125, "grad_norm": 0.3152504563331604, "learning_rate": 9.729718877603861e-05, "loss": 0.015125786885619164, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01524, "step": 80, "tokens/total": 2422240, "tokens/train_per_sec_per_gpu": 38.73, "tokens/trainable": 36522} +{"epoch": 0.31640625, "grad_norm": 0.6696128249168396, "learning_rate": 9.719721959634592e-05, "loss": 0.023452740162611008, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.02373, "step": 81, "tokens/total": 2452560, "tokens/train_per_sec_per_gpu": 33.67, "tokens/trainable": 36979} +{"epoch": 0.3203125, "grad_norm": 0.3085023760795593, "learning_rate": 9.709549441810375e-05, "loss": 0.011503729969263077, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.44, "memory/max_allocated (GiB)": 33.44, "ppl": 1.01157, "step": 82, "tokens/total": 2481024, "tokens/train_per_sec_per_gpu": 33.93, "tokens/trainable": 37439} +{"epoch": 0.32421875, "grad_norm": 0.44436803460121155, "learning_rate": 9.699201747451195e-05, "loss": 0.01643884740769863, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01657, "step": 83, "tokens/total": 2511296, "tokens/train_per_sec_per_gpu": 33.54, "tokens/trainable": 37892} +{"epoch": 0.328125, "grad_norm": 1.1241978406906128, "learning_rate": 9.688679307166854e-05, "loss": 0.0325784869492054, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.03311, "step": 84, "tokens/total": 2541792, "tokens/train_per_sec_per_gpu": 33.19, "tokens/trainable": 38355} +{"epoch": 0.33203125, "grad_norm": 0.493679016828537, "learning_rate": 9.677982558839042e-05, "loss": 0.0119265615940094, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.012, "step": 85, "tokens/total": 2572032, "tokens/train_per_sec_per_gpu": 34.07, "tokens/trainable": 38810} +{"epoch": 0.3359375, "grad_norm": 1.3241630792617798, "learning_rate": 9.66711194760312e-05, "loss": 0.04201122000813484, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.04291, "step": 86, "tokens/total": 2602512, "tokens/train_per_sec_per_gpu": 30.89, "tokens/trainable": 39240} +{"epoch": 0.33984375, "grad_norm": 0.4352894723415375, "learning_rate": 9.656067925829593e-05, "loss": 0.028608884662389755, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.02902, "step": 87, "tokens/total": 2632880, "tokens/train_per_sec_per_gpu": 37.0, "tokens/trainable": 39748} +{"epoch": 0.34375, "grad_norm": 0.48992764949798584, "learning_rate": 9.644850953105288e-05, "loss": 0.035998307168483734, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.03665, "step": 88, "tokens/total": 2663040, "tokens/train_per_sec_per_gpu": 33.35, "tokens/trainable": 40188} +{"epoch": 0.34765625, "grad_norm": 0.491361141204834, "learning_rate": 9.633461496214225e-05, "loss": 0.02381829358637333, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.0241, "step": 89, "tokens/total": 2693504, "tokens/train_per_sec_per_gpu": 33.36, "tokens/trainable": 40632} +{"epoch": 0.3515625, "grad_norm": 0.4422471225261688, "learning_rate": 9.621900029118195e-05, "loss": 0.022936182096600533, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0232, "step": 90, "tokens/total": 2723792, "tokens/train_per_sec_per_gpu": 31.37, "tokens/trainable": 41050} +{"epoch": 0.35546875, "grad_norm": 0.4762331247329712, "learning_rate": 9.610167032937036e-05, "loss": 0.017587462440133095, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.01774, "step": 91, "tokens/total": 2754048, "tokens/train_per_sec_per_gpu": 37.05, "tokens/trainable": 41515} +{"epoch": 0.359375, "grad_norm": 0.4137004017829895, "learning_rate": 9.598262995928611e-05, "loss": 0.03153759241104126, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.03204, "step": 92, "tokens/total": 2784480, "tokens/train_per_sec_per_gpu": 36.23, "tokens/trainable": 41972} +{"epoch": 0.36328125, "grad_norm": 0.5270156264305115, "learning_rate": 9.586188413468492e-05, "loss": 0.023751404136419296, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.02404, "step": 93, "tokens/total": 2814928, "tokens/train_per_sec_per_gpu": 35.43, "tokens/trainable": 42442} +{"epoch": 0.3671875, "grad_norm": 0.30221623182296753, "learning_rate": 9.57394378802934e-05, "loss": 0.012607569806277752, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.01269, "step": 94, "tokens/total": 2845408, "tokens/train_per_sec_per_gpu": 37.18, "tokens/trainable": 42928} +{"epoch": 0.37109375, "grad_norm": 0.34380587935447693, "learning_rate": 9.56152962916e-05, "loss": 0.021355444565415382, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.02159, "step": 95, "tokens/total": 2875760, "tokens/train_per_sec_per_gpu": 39.28, "tokens/trainable": 43426} +{"epoch": 0.375, "grad_norm": 0.5729417204856873, "learning_rate": 9.548946453464296e-05, "loss": 0.011794034391641617, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01186, "step": 96, "tokens/total": 2906096, "tokens/train_per_sec_per_gpu": 33.52, "tokens/trainable": 43869} +{"epoch": 0.37890625, "grad_norm": 0.1258164346218109, "learning_rate": 9.53619478457953e-05, "loss": 0.002208392135798931, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00221, "step": 97, "tokens/total": 2936560, "tokens/train_per_sec_per_gpu": 32.02, "tokens/trainable": 44300} +{"epoch": 0.3828125, "grad_norm": 0.43118366599082947, "learning_rate": 9.523275153154695e-05, "loss": 0.015600357204675674, "memory/device_reserved (GiB)": 35.19, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01572, "step": 98, "tokens/total": 2966880, "tokens/train_per_sec_per_gpu": 39.37, "tokens/trainable": 44791} +{"epoch": 0.38671875, "grad_norm": 0.505718469619751, "learning_rate": 9.51018809682839e-05, "loss": 0.011874521151185036, "memory/device_reserved (GiB)": 35.76, "memory/max_active (GiB)": 34.03, "memory/max_allocated (GiB)": 34.03, "ppl": 1.01195, "step": 99, "tokens/total": 2997472, "tokens/train_per_sec_per_gpu": 36.71, "tokens/trainable": 45254} +{"epoch": 0.390625, "grad_norm": 3.060197591781616, "learning_rate": 9.49693416020645e-05, "loss": 0.014575858600437641, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01468, "step": 100, "tokens/total": 3027840, "tokens/train_per_sec_per_gpu": 33.57, "tokens/trainable": 45698} +{"epoch": 0.39453125, "grad_norm": 0.38254642486572266, "learning_rate": 9.483513894839276e-05, "loss": 0.009431221522390842, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00948, "step": 101, "tokens/total": 3058080, "tokens/train_per_sec_per_gpu": 33.57, "tokens/trainable": 46141} +{"epoch": 0.3984375, "grad_norm": 0.8782851099967957, "learning_rate": 9.469927859198888e-05, "loss": 0.05128197371959686, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.05262, "step": 102, "tokens/total": 3088656, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 46598} +{"epoch": 0.40234375, "grad_norm": 0.42221176624298096, "learning_rate": 9.456176618655689e-05, "loss": 0.009157870896160603, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.0092, "step": 103, "tokens/total": 3119232, "tokens/train_per_sec_per_gpu": 32.45, "tokens/trainable": 47065} +{"epoch": 0.40625, "grad_norm": 1.2625210285186768, "learning_rate": 9.442260745454927e-05, "loss": 0.01384878158569336, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01395, "step": 104, "tokens/total": 3149504, "tokens/train_per_sec_per_gpu": 38.29, "tokens/trainable": 47554} +{"epoch": 0.41015625, "grad_norm": 0.3072832226753235, "learning_rate": 9.428180818692884e-05, "loss": 0.00823851116001606, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00827, "step": 105, "tokens/total": 3179872, "tokens/train_per_sec_per_gpu": 36.88, "tokens/trainable": 48047} +{"epoch": 0.4140625, "grad_norm": 0.7415916323661804, "learning_rate": 9.413937424292791e-05, "loss": 0.03210819885134697, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.03263, "step": 106, "tokens/total": 3208128, "tokens/train_per_sec_per_gpu": 34.62, "tokens/trainable": 48498} +{"epoch": 0.41796875, "grad_norm": 0.25024500489234924, "learning_rate": 9.399531154980424e-05, "loss": 0.00918180774897337, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00922, "step": 107, "tokens/total": 3238224, "tokens/train_per_sec_per_gpu": 31.18, "tokens/trainable": 48933} +{"epoch": 0.421875, "grad_norm": 0.5457918047904968, "learning_rate": 9.384962610259455e-05, "loss": 0.015052877366542816, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.01517, "step": 108, "tokens/total": 3268592, "tokens/train_per_sec_per_gpu": 38.48, "tokens/trainable": 49404} +{"epoch": 0.42578125, "grad_norm": 0.2508637607097626, "learning_rate": 9.370232396386494e-05, "loss": 0.007351120002567768, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00738, "step": 109, "tokens/total": 3298720, "tokens/train_per_sec_per_gpu": 37.32, "tokens/trainable": 49871} +{"epoch": 0.4296875, "grad_norm": 0.2964157462120056, "learning_rate": 9.355341126345868e-05, "loss": 0.014206478372216225, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.01431, "step": 110, "tokens/total": 3329040, "tokens/train_per_sec_per_gpu": 33.58, "tokens/trainable": 50311} +{"epoch": 0.43359375, "grad_norm": 0.3096546232700348, "learning_rate": 9.340289419824107e-05, "loss": 0.010700306855142117, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01076, "step": 111, "tokens/total": 3359248, "tokens/train_per_sec_per_gpu": 33.86, "tokens/trainable": 50754} +{"epoch": 0.4375, "grad_norm": 0.4973372220993042, "learning_rate": 9.325077903184159e-05, "loss": 0.013185497373342514, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.01327, "step": 112, "tokens/total": 3389280, "tokens/train_per_sec_per_gpu": 32.1, "tokens/trainable": 51200} +{"epoch": 0.44140625, "grad_norm": 0.4105257987976074, "learning_rate": 9.30970720943932e-05, "loss": 0.0069708251394331455, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.007, "step": 113, "tokens/total": 3419728, "tokens/train_per_sec_per_gpu": 33.22, "tokens/trainable": 51656} +{"epoch": 0.4453125, "grad_norm": 0.2707289159297943, "learning_rate": 9.2941779782269e-05, "loss": 0.0033873419743031263, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00339, "step": 114, "tokens/total": 3449984, "tokens/train_per_sec_per_gpu": 30.92, "tokens/trainable": 52069} +{"epoch": 0.44921875, "grad_norm": 0.5602349042892456, "learning_rate": 9.278490855781596e-05, "loss": 0.00848664902150631, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00852, "step": 115, "tokens/total": 3480352, "tokens/train_per_sec_per_gpu": 36.94, "tokens/trainable": 52559} +{"epoch": 0.453125, "grad_norm": 0.5224027037620544, "learning_rate": 9.262646494908604e-05, "loss": 0.02242768369615078, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.02268, "step": 116, "tokens/total": 3510656, "tokens/train_per_sec_per_gpu": 39.22, "tokens/trainable": 53061} +{"epoch": 0.45703125, "grad_norm": 1.2213423252105713, "learning_rate": 9.246645554956457e-05, "loss": 0.020464815199375153, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.02068, "step": 117, "tokens/total": 3541104, "tokens/train_per_sec_per_gpu": 35.89, "tokens/trainable": 53520} +{"epoch": 0.4609375, "grad_norm": 1.30115807056427, "learning_rate": 9.230488701789578e-05, "loss": 0.025597743690013885, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.02593, "step": 118, "tokens/total": 3571744, "tokens/train_per_sec_per_gpu": 33.36, "tokens/trainable": 53976} +{"epoch": 0.46484375, "grad_norm": 1.0988223552703857, "learning_rate": 9.214176607760577e-05, "loss": 0.02888210117816925, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.0293, "step": 119, "tokens/total": 3601808, "tokens/train_per_sec_per_gpu": 34.93, "tokens/trainable": 54408} +{"epoch": 0.46875, "grad_norm": 0.3977462649345398, "learning_rate": 9.197709951682268e-05, "loss": 0.01678757183253765, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01693, "step": 120, "tokens/total": 3632096, "tokens/train_per_sec_per_gpu": 37.14, "tokens/trainable": 54884} +{"epoch": 0.47265625, "grad_norm": 0.4872536063194275, "learning_rate": 9.181089418799428e-05, "loss": 0.022638533264398575, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0229, "step": 121, "tokens/total": 3662416, "tokens/train_per_sec_per_gpu": 33.15, "tokens/trainable": 55303} +{"epoch": 0.4765625, "grad_norm": 0.4663882255554199, "learning_rate": 9.164315700760271e-05, "loss": 0.023266036063432693, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.02354, "step": 122, "tokens/total": 3692992, "tokens/train_per_sec_per_gpu": 31.94, "tokens/trainable": 55743} +{"epoch": 0.48046875, "grad_norm": 0.230746328830719, "learning_rate": 9.147389495587671e-05, "loss": 0.007810275070369244, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00784, "step": 123, "tokens/total": 3722992, "tokens/train_per_sec_per_gpu": 33.95, "tokens/trainable": 56187} +{"epoch": 0.484375, "grad_norm": 0.17318949103355408, "learning_rate": 9.130311507650116e-05, "loss": 0.006973203271627426, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.007, "step": 124, "tokens/total": 3753184, "tokens/train_per_sec_per_gpu": 33.63, "tokens/trainable": 56632} +{"epoch": 0.48828125, "grad_norm": 0.19994892179965973, "learning_rate": 9.113082447632394e-05, "loss": 0.0068689389154314995, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00689, "step": 125, "tokens/total": 3783440, "tokens/train_per_sec_per_gpu": 37.26, "tokens/trainable": 57124} +{"epoch": 0.4921875, "grad_norm": 0.11329996585845947, "learning_rate": 9.09570303250602e-05, "loss": 0.004266452044248581, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00428, "step": 126, "tokens/total": 3813840, "tokens/train_per_sec_per_gpu": 33.38, "tokens/trainable": 57570} +{"epoch": 0.49609375, "grad_norm": 0.4156385362148285, "learning_rate": 9.078173985499394e-05, "loss": 0.02041183039546013, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.02062, "step": 127, "tokens/total": 3844464, "tokens/train_per_sec_per_gpu": 35.36, "tokens/trainable": 58048} +{"epoch": 0.5, "grad_norm": 0.09819997847080231, "learning_rate": 9.060496036067713e-05, "loss": 0.0031540761701762676, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.00316, "step": 128, "tokens/total": 3872816, "tokens/train_per_sec_per_gpu": 35.79, "tokens/trainable": 58497} +{"epoch": 0.50390625, "grad_norm": 0.34266355633735657, "learning_rate": 9.042669919862615e-05, "loss": 0.017772674560546875, "memory/device_reserved (GiB)": 36.0, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.01793, "step": 129, "tokens/total": 3902992, "tokens/train_per_sec_per_gpu": 33.44, "tokens/trainable": 58947} +{"epoch": 0.5078125, "grad_norm": 0.3253064751625061, "learning_rate": 9.024696378701557e-05, "loss": 0.018172409385442734, "memory/device_reserved (GiB)": 34.74, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01834, "step": 130, "tokens/total": 3933456, "tokens/train_per_sec_per_gpu": 37.56, "tokens/trainable": 59415} +{"epoch": 0.51171875, "grad_norm": 0.3167641758918762, "learning_rate": 9.006576160536948e-05, "loss": 0.008536357432603836, "memory/device_reserved (GiB)": 34.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00857, "step": 131, "tokens/total": 3963536, "tokens/train_per_sec_per_gpu": 32.75, "tokens/trainable": 59851} +{"epoch": 0.515625, "grad_norm": 0.12403249740600586, "learning_rate": 8.988310019425035e-05, "loss": 0.0019342785235494375, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00194, "step": 132, "tokens/total": 3994000, "tokens/train_per_sec_per_gpu": 36.25, "tokens/trainable": 60325} +{"epoch": 0.51953125, "grad_norm": 0.25862130522727966, "learning_rate": 8.969898715494506e-05, "loss": 0.01549853477627039, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01562, "step": 133, "tokens/total": 4024352, "tokens/train_per_sec_per_gpu": 35.97, "tokens/trainable": 60818} +{"epoch": 0.5234375, "grad_norm": 0.3527744710445404, "learning_rate": 8.951343014914869e-05, "loss": 0.006827985402196646, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.39, "memory/max_allocated (GiB)": 33.39, "ppl": 1.00685, "step": 134, "tokens/total": 4052832, "tokens/train_per_sec_per_gpu": 32.28, "tokens/trainable": 61263} +{"epoch": 0.52734375, "grad_norm": 0.3797135353088379, "learning_rate": 8.932643689864568e-05, "loss": 0.010647830553352833, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.0107, "step": 135, "tokens/total": 4083216, "tokens/train_per_sec_per_gpu": 36.58, "tokens/trainable": 61734} +{"epoch": 0.53125, "grad_norm": 0.3589562177658081, "learning_rate": 8.913801518498845e-05, "loss": 0.022732166573405266, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.7, "memory/max_allocated (GiB)": 33.7, "ppl": 1.02299, "step": 136, "tokens/total": 4113392, "tokens/train_per_sec_per_gpu": 38.83, "tokens/trainable": 62210} +{"epoch": 0.53515625, "grad_norm": 0.34385553002357483, "learning_rate": 8.894817284917364e-05, "loss": 0.017764274030923843, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01792, "step": 137, "tokens/total": 4143696, "tokens/train_per_sec_per_gpu": 36.87, "tokens/trainable": 62707} +{"epoch": 0.5390625, "grad_norm": 0.24294431507587433, "learning_rate": 8.875691779131569e-05, "loss": 0.01984233781695366, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.02004, "step": 138, "tokens/total": 4174000, "tokens/train_per_sec_per_gpu": 37.17, "tokens/trainable": 63166} +{"epoch": 0.54296875, "grad_norm": 0.6662021279335022, "learning_rate": 8.856425797031829e-05, "loss": 0.012637479230761528, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.01272, "step": 139, "tokens/total": 4204432, "tokens/train_per_sec_per_gpu": 36.96, "tokens/trainable": 63646} +{"epoch": 0.546875, "grad_norm": 0.26386311650276184, "learning_rate": 8.837020140354295e-05, "loss": 0.014607764780521393, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.42, "memory/max_allocated (GiB)": 33.42, "ppl": 1.01471, "step": 140, "tokens/total": 4232752, "tokens/train_per_sec_per_gpu": 33.6, "tokens/trainable": 64063} +{"epoch": 0.55078125, "grad_norm": 0.2475082129240036, "learning_rate": 8.817475616647554e-05, "loss": 0.012658985331654549, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.01274, "step": 141, "tokens/total": 4262976, "tokens/train_per_sec_per_gpu": 35.67, "tokens/trainable": 64520} +{"epoch": 0.5546875, "grad_norm": 0.21145126223564148, "learning_rate": 8.797793039239017e-05, "loss": 0.009737009182572365, "memory/device_reserved (GiB)": 35.79, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00978, "step": 142, "tokens/total": 4293520, "tokens/train_per_sec_per_gpu": 34.48, "tokens/trainable": 64958} +{"epoch": 0.55859375, "grad_norm": 0.2296362668275833, "learning_rate": 8.777973227201069e-05, "loss": 0.013068155385553837, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01315, "step": 143, "tokens/total": 4323520, "tokens/train_per_sec_per_gpu": 38.3, "tokens/trainable": 65437} +{"epoch": 0.5625, "grad_norm": 0.2896386981010437, "learning_rate": 8.758017005316988e-05, "loss": 0.010017371736466885, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01007, "step": 144, "tokens/total": 4353616, "tokens/train_per_sec_per_gpu": 37.25, "tokens/trainable": 65903} +{"epoch": 0.56640625, "grad_norm": 0.41864296793937683, "learning_rate": 8.737925204046629e-05, "loss": 0.025722038000822067, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.02606, "step": 145, "tokens/total": 4383856, "tokens/train_per_sec_per_gpu": 32.08, "tokens/trainable": 66334} +{"epoch": 0.5703125, "grad_norm": 0.27801141142845154, "learning_rate": 8.717698659491851e-05, "loss": 0.017308663576841354, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.01746, "step": 146, "tokens/total": 4414128, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 66785} +{"epoch": 0.57421875, "grad_norm": 0.575922966003418, "learning_rate": 8.697338213361735e-05, "loss": 0.017252381891012192, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0174, "step": 147, "tokens/total": 4444384, "tokens/train_per_sec_per_gpu": 35.81, "tokens/trainable": 67199} +{"epoch": 0.578125, "grad_norm": 0.4430011510848999, "learning_rate": 8.676844712937552e-05, "loss": 0.022439993917942047, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.02269, "step": 148, "tokens/total": 4474912, "tokens/train_per_sec_per_gpu": 35.64, "tokens/trainable": 67683} +{"epoch": 0.58203125, "grad_norm": 0.26630517840385437, "learning_rate": 8.656219011037509e-05, "loss": 0.010438371449708939, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01049, "step": 149, "tokens/total": 4505328, "tokens/train_per_sec_per_gpu": 34.4, "tokens/trainable": 68127} +{"epoch": 0.5859375, "grad_norm": 0.15251637995243073, "learning_rate": 8.63546196598125e-05, "loss": 0.004219424910843372, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00423, "step": 150, "tokens/total": 4535424, "tokens/train_per_sec_per_gpu": 33.75, "tokens/trainable": 68595} +{"epoch": 0.58984375, "grad_norm": 0.21238818764686584, "learning_rate": 8.614574441554145e-05, "loss": 0.01187204197049141, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01194, "step": 151, "tokens/total": 4565536, "tokens/train_per_sec_per_gpu": 30.68, "tokens/trainable": 69062} +{"epoch": 0.59375, "grad_norm": 0.5052844285964966, "learning_rate": 8.593557306971349e-05, "loss": 0.007624611258506775, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00765, "step": 152, "tokens/total": 4596048, "tokens/train_per_sec_per_gpu": 35.96, "tokens/trainable": 69509} +{"epoch": 0.59765625, "grad_norm": 0.46278703212738037, "learning_rate": 8.572411436841618e-05, "loss": 0.01224217563867569, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.01232, "step": 153, "tokens/total": 4626480, "tokens/train_per_sec_per_gpu": 34.45, "tokens/trainable": 69972} +{"epoch": 0.6015625, "grad_norm": 0.4227867126464844, "learning_rate": 8.551137711130922e-05, "loss": 0.0060340641066432, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00605, "step": 154, "tokens/total": 4656800, "tokens/train_per_sec_per_gpu": 37.3, "tokens/trainable": 70439} +{"epoch": 0.60546875, "grad_norm": 0.5990466475486755, "learning_rate": 8.529737015125824e-05, "loss": 0.018272625282406807, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01844, "step": 155, "tokens/total": 4685120, "tokens/train_per_sec_per_gpu": 37.43, "tokens/trainable": 70881} +{"epoch": 0.609375, "grad_norm": 0.6471202969551086, "learning_rate": 8.508210239396639e-05, "loss": 0.01053079217672348, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01059, "step": 156, "tokens/total": 4715664, "tokens/train_per_sec_per_gpu": 34.48, "tokens/trainable": 71339} +{"epoch": 0.61328125, "grad_norm": 0.00898966658860445, "learning_rate": 8.486558279760375e-05, "loss": 0.0001828348613344133, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00018, "step": 157, "tokens/total": 4745824, "tokens/train_per_sec_per_gpu": 35.55, "tokens/trainable": 71801} +{"epoch": 0.6171875, "grad_norm": 0.2965949773788452, "learning_rate": 8.464782037243449e-05, "loss": 0.014287484809756279, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01439, "step": 158, "tokens/total": 4776224, "tokens/train_per_sec_per_gpu": 32.65, "tokens/trainable": 72237} +{"epoch": 0.62109375, "grad_norm": 0.3070797324180603, "learning_rate": 8.442882418044202e-05, "loss": 0.025281894952058792, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.0256, "step": 159, "tokens/total": 4806304, "tokens/train_per_sec_per_gpu": 31.03, "tokens/trainable": 72664} +{"epoch": 0.625, "grad_norm": 0.29432007670402527, "learning_rate": 8.420860333495179e-05, "loss": 0.005029833409935236, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00504, "step": 160, "tokens/total": 4836688, "tokens/train_per_sec_per_gpu": 32.1, "tokens/trainable": 73119} +{"epoch": 0.62890625, "grad_norm": 0.38903817534446716, "learning_rate": 8.398716700025208e-05, "loss": 0.010762909427285194, "memory/device_reserved (GiB)": 37.89, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.01082, "step": 161, "tokens/total": 4867008, "tokens/train_per_sec_per_gpu": 31.51, "tokens/trainable": 73540} +{"epoch": 0.6328125, "grad_norm": 0.23661470413208008, "learning_rate": 8.376452439121266e-05, "loss": 0.003404806135222316, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00341, "step": 162, "tokens/total": 4897584, "tokens/train_per_sec_per_gpu": 33.18, "tokens/trainable": 73990} +{"epoch": 0.63671875, "grad_norm": 0.23111319541931152, "learning_rate": 8.354068477290124e-05, "loss": 0.0172940231859684, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01744, "step": 163, "tokens/total": 4927904, "tokens/train_per_sec_per_gpu": 35.89, "tokens/trainable": 74454} +{"epoch": 0.640625, "grad_norm": 0.7157747745513916, "learning_rate": 8.331565746019807e-05, "loss": 0.025878142565488815, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.02622, "step": 164, "tokens/total": 4958496, "tokens/train_per_sec_per_gpu": 33.96, "tokens/trainable": 74905} +{"epoch": 0.64453125, "grad_norm": 0.3431304097175598, "learning_rate": 8.308945181740812e-05, "loss": 0.003934288397431374, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00394, "step": 165, "tokens/total": 4988832, "tokens/train_per_sec_per_gpu": 31.03, "tokens/trainable": 75329} +{"epoch": 0.6484375, "grad_norm": 0.0704391747713089, "learning_rate": 8.286207725787153e-05, "loss": 0.0012416213285177946, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00124, "step": 166, "tokens/total": 5019136, "tokens/train_per_sec_per_gpu": 32.34, "tokens/trainable": 75770} +{"epoch": 0.65234375, "grad_norm": 0.6784317493438721, "learning_rate": 8.263354324357182e-05, "loss": 0.008922886103391647, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00896, "step": 167, "tokens/total": 5049808, "tokens/train_per_sec_per_gpu": 31.81, "tokens/trainable": 76231} +{"epoch": 0.65625, "grad_norm": 0.41060712933540344, "learning_rate": 8.240385928474219e-05, "loss": 0.018168501555919647, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.01833, "step": 168, "tokens/total": 5080320, "tokens/train_per_sec_per_gpu": 33.83, "tokens/trainable": 76682} +{"epoch": 0.66015625, "grad_norm": 0.13889877498149872, "learning_rate": 8.217303493946967e-05, "loss": 0.0031139992643147707, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00312, "step": 169, "tokens/total": 5110656, "tokens/train_per_sec_per_gpu": 34.87, "tokens/trainable": 77114} +{"epoch": 0.6640625, "grad_norm": 0.6820456385612488, "learning_rate": 8.194107981329746e-05, "loss": 0.030903562903404236, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.03139, "step": 170, "tokens/total": 5141056, "tokens/train_per_sec_per_gpu": 34.61, "tokens/trainable": 77563} +{"epoch": 0.66796875, "grad_norm": 0.20577676594257355, "learning_rate": 8.170800355882518e-05, "loss": 0.0030413041822612286, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00305, "step": 171, "tokens/total": 5171312, "tokens/train_per_sec_per_gpu": 33.7, "tokens/trainable": 78023} +{"epoch": 0.671875, "grad_norm": 0.2194790244102478, "learning_rate": 8.147381587530713e-05, "loss": 0.004004347138106823, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00401, "step": 172, "tokens/total": 5201744, "tokens/train_per_sec_per_gpu": 30.23, "tokens/trainable": 78477} +{"epoch": 0.67578125, "grad_norm": 0.5019084811210632, "learning_rate": 8.123852650824877e-05, "loss": 0.006000855006277561, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00602, "step": 173, "tokens/total": 5232176, "tokens/train_per_sec_per_gpu": 33.51, "tokens/trainable": 78924} +{"epoch": 0.6796875, "grad_norm": 0.36753812432289124, "learning_rate": 8.100214524900103e-05, "loss": 0.004308007657527924, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.62, "memory/max_allocated (GiB)": 33.62, "ppl": 1.00432, "step": 174, "tokens/total": 5261904, "tokens/train_per_sec_per_gpu": 30.59, "tokens/trainable": 79342} +{"epoch": 0.68359375, "grad_norm": 0.42971113324165344, "learning_rate": 8.076468193435301e-05, "loss": 0.0007982449606060982, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.0008, "step": 175, "tokens/total": 5292192, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 79812} +{"epoch": 0.6875, "grad_norm": 0.7103844285011292, "learning_rate": 8.052614644612253e-05, "loss": 0.005436811596155167, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00545, "step": 176, "tokens/total": 5322704, "tokens/train_per_sec_per_gpu": 36.73, "tokens/trainable": 80290} +{"epoch": 0.69140625, "grad_norm": 0.5654855966567993, "learning_rate": 8.028654871074489e-05, "loss": 0.007358669303357601, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00739, "step": 177, "tokens/total": 5353040, "tokens/train_per_sec_per_gpu": 33.08, "tokens/trainable": 80737} +{"epoch": 0.6953125, "grad_norm": 0.035665832459926605, "learning_rate": 8.004589869885986e-05, "loss": 0.0001784512132871896, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00018, "step": 178, "tokens/total": 5383248, "tokens/train_per_sec_per_gpu": 35.33, "tokens/trainable": 81205} +{"epoch": 0.69921875, "grad_norm": 0.01852536015212536, "learning_rate": 7.980420642489674e-05, "loss": 0.0002031420881394297, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0002, "step": 179, "tokens/total": 5413248, "tokens/train_per_sec_per_gpu": 33.53, "tokens/trainable": 81652} +{"epoch": 0.703125, "grad_norm": 0.10783406347036362, "learning_rate": 7.95614819466576e-05, "loss": 0.0007448110263794661, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00075, "step": 180, "tokens/total": 5443424, "tokens/train_per_sec_per_gpu": 35.07, "tokens/trainable": 82125} +{"epoch": 0.70703125, "grad_norm": 0.15577788650989532, "learning_rate": 7.931773536489872e-05, "loss": 0.000892514712177217, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00089, "step": 181, "tokens/total": 5473680, "tokens/train_per_sec_per_gpu": 39.35, "tokens/trainable": 82642} +{"epoch": 0.7109375, "grad_norm": 0.7524833679199219, "learning_rate": 7.907297682291035e-05, "loss": 0.01774725317955017, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01791, "step": 182, "tokens/total": 5504032, "tokens/train_per_sec_per_gpu": 34.47, "tokens/trainable": 83123} +{"epoch": 0.71484375, "grad_norm": 0.8974100947380066, "learning_rate": 7.882721650609442e-05, "loss": 0.016709204763174057, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.01685, "step": 183, "tokens/total": 5534224, "tokens/train_per_sec_per_gpu": 33.26, "tokens/trainable": 83570} +{"epoch": 0.71875, "grad_norm": 0.31876227259635925, "learning_rate": 7.85804646415409e-05, "loss": 0.0018143621273338795, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00182, "step": 184, "tokens/total": 5564704, "tokens/train_per_sec_per_gpu": 36.04, "tokens/trainable": 84076} +{"epoch": 0.72265625, "grad_norm": 0.3767867684364319, "learning_rate": 7.833273149760207e-05, "loss": 0.030571268871426582, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.03104, "step": 185, "tokens/total": 5594912, "tokens/train_per_sec_per_gpu": 34.97, "tokens/trainable": 84551} +{"epoch": 0.7265625, "grad_norm": 0.025328254327178, "learning_rate": 7.808402738346527e-05, "loss": 0.0003344593569636345, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00033, "step": 186, "tokens/total": 5625280, "tokens/train_per_sec_per_gpu": 33.92, "tokens/trainable": 85012} +{"epoch": 0.73046875, "grad_norm": 0.37010785937309265, "learning_rate": 7.783436264872382e-05, "loss": 0.009675242006778717, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00972, "step": 187, "tokens/total": 5655568, "tokens/train_per_sec_per_gpu": 32.23, "tokens/trainable": 85452} +{"epoch": 0.734375, "grad_norm": 0.32615038752555847, "learning_rate": 7.758374768294647e-05, "loss": 0.010340536944568157, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01039, "step": 188, "tokens/total": 5685824, "tokens/train_per_sec_per_gpu": 34.49, "tokens/trainable": 85942} +{"epoch": 0.73828125, "grad_norm": 0.6183643937110901, "learning_rate": 7.733219291524489e-05, "loss": 0.00930570624768734, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00935, "step": 189, "tokens/total": 5716192, "tokens/train_per_sec_per_gpu": 36.56, "tokens/trainable": 86457} +{"epoch": 0.7421875, "grad_norm": 0.31889259815216064, "learning_rate": 7.707970881383977e-05, "loss": 0.004300425294786692, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00431, "step": 190, "tokens/total": 5746784, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 86928} +{"epoch": 0.74609375, "grad_norm": 0.34553802013397217, "learning_rate": 7.682630588562518e-05, "loss": 0.00664468202739954, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00667, "step": 191, "tokens/total": 5777520, "tokens/train_per_sec_per_gpu": 36.31, "tokens/trainable": 87384} +{"epoch": 0.75, "grad_norm": 0.2645066976547241, "learning_rate": 7.657199467573129e-05, "loss": 0.006615218240767717, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00664, "step": 192, "tokens/total": 5807952, "tokens/train_per_sec_per_gpu": 32.03, "tokens/trainable": 87814} +{"epoch": 0.75390625, "grad_norm": 0.3881167471408844, "learning_rate": 7.631678576708561e-05, "loss": 0.014919335022568703, "memory/device_reserved (GiB)": 36.02, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01503, "step": 193, "tokens/total": 5838384, "tokens/train_per_sec_per_gpu": 35.64, "tokens/trainable": 88297} +{"epoch": 0.7578125, "grad_norm": 0.6184028387069702, "learning_rate": 7.606068977997255e-05, "loss": 0.026482408866286278, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.02684, "step": 194, "tokens/total": 5868704, "tokens/train_per_sec_per_gpu": 35.37, "tokens/trainable": 88768} +{"epoch": 0.76171875, "grad_norm": 0.17681379616260529, "learning_rate": 7.580371737159148e-05, "loss": 0.0036258562467992306, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.38, "memory/max_allocated (GiB)": 33.38, "ppl": 1.00363, "step": 195, "tokens/total": 5896928, "tokens/train_per_sec_per_gpu": 30.88, "tokens/trainable": 89201} +{"epoch": 0.765625, "grad_norm": 0.09678306430578232, "learning_rate": 7.554587923561324e-05, "loss": 0.0024895307142287493, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00249, "step": 196, "tokens/total": 5927328, "tokens/train_per_sec_per_gpu": 37.84, "tokens/trainable": 89663} +{"epoch": 0.76953125, "grad_norm": 0.033167868852615356, "learning_rate": 7.528718610173511e-05, "loss": 0.0012309665326029062, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.34, "memory/max_allocated (GiB)": 33.34, "ppl": 1.00123, "step": 197, "tokens/total": 5955648, "tokens/train_per_sec_per_gpu": 38.68, "tokens/trainable": 90111} +{"epoch": 0.7734375, "grad_norm": 0.36096563935279846, "learning_rate": 7.502764873523431e-05, "loss": 0.011163209564983845, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.01123, "step": 198, "tokens/total": 5985904, "tokens/train_per_sec_per_gpu": 34.73, "tokens/trainable": 90572} +{"epoch": 0.77734375, "grad_norm": 0.07940459251403809, "learning_rate": 7.476727793652011e-05, "loss": 0.00193371856585145, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 34.0, "memory/max_allocated (GiB)": 34.0, "ppl": 1.00194, "step": 199, "tokens/total": 6016656, "tokens/train_per_sec_per_gpu": 37.02, "tokens/trainable": 91051} +{"epoch": 0.78125, "grad_norm": 0.28284725546836853, "learning_rate": 7.450608454068415e-05, "loss": 0.008390676230192184, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00843, "step": 200, "tokens/total": 6046912, "tokens/train_per_sec_per_gpu": 38.74, "tokens/trainable": 91527} +{"epoch": 0.78515625, "grad_norm": 0.02596334181725979, "learning_rate": 7.424407941704987e-05, "loss": 0.0006643222877755761, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00066, "step": 201, "tokens/total": 6076912, "tokens/train_per_sec_per_gpu": 34.77, "tokens/trainable": 91973} +{"epoch": 0.7890625, "grad_norm": 0.26352518796920776, "learning_rate": 7.398127346871986e-05, "loss": 0.01052158698439598, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.01058, "step": 202, "tokens/total": 6107264, "tokens/train_per_sec_per_gpu": 37.23, "tokens/trainable": 92438} +{"epoch": 0.79296875, "grad_norm": 0.4626573920249939, "learning_rate": 7.371767763212238e-05, "loss": 0.026418011635541916, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.02677, "step": 203, "tokens/total": 6137344, "tokens/train_per_sec_per_gpu": 37.12, "tokens/trainable": 92886} +{"epoch": 0.796875, "grad_norm": 0.49221205711364746, "learning_rate": 7.345330287655617e-05, "loss": 0.005499291233718395, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00551, "step": 204, "tokens/total": 6167888, "tokens/train_per_sec_per_gpu": 34.54, "tokens/trainable": 93359} +{"epoch": 0.80078125, "grad_norm": 0.14302127063274384, "learning_rate": 7.31881602037339e-05, "loss": 0.002654140582308173, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00266, "step": 205, "tokens/total": 6198080, "tokens/train_per_sec_per_gpu": 33.97, "tokens/trainable": 93783} +{"epoch": 0.8046875, "grad_norm": 0.1867009699344635, "learning_rate": 7.29222606473245e-05, "loss": 0.005218994803726673, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00523, "step": 206, "tokens/total": 6228480, "tokens/train_per_sec_per_gpu": 37.94, "tokens/trainable": 94262} +{"epoch": 0.80859375, "grad_norm": 0.396358460187912, "learning_rate": 7.265561527249383e-05, "loss": 0.010595105588436127, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01065, "step": 207, "tokens/total": 6258640, "tokens/train_per_sec_per_gpu": 36.03, "tokens/trainable": 94753} +{"epoch": 0.8125, "grad_norm": 0.04140181094408035, "learning_rate": 7.238823517544436e-05, "loss": 0.0008800303330644965, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00088, "step": 208, "tokens/total": 6288928, "tokens/train_per_sec_per_gpu": 34.21, "tokens/trainable": 95238} +{"epoch": 0.81640625, "grad_norm": 0.09654782712459564, "learning_rate": 7.212013148295333e-05, "loss": 0.0018917301204055548, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00189, "step": 209, "tokens/total": 6319568, "tokens/train_per_sec_per_gpu": 29.65, "tokens/trainable": 95639} +{"epoch": 0.8203125, "grad_norm": 0.2853289842605591, "learning_rate": 7.185131535190975e-05, "loss": 0.005200013052672148, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00521, "step": 210, "tokens/total": 6349824, "tokens/train_per_sec_per_gpu": 35.91, "tokens/trainable": 96096} +{"epoch": 0.82421875, "grad_norm": 1.9312809705734253, "learning_rate": 7.158179796885005e-05, "loss": 0.012804300524294376, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01289, "step": 211, "tokens/total": 6379920, "tokens/train_per_sec_per_gpu": 34.64, "tokens/trainable": 96517} +{"epoch": 0.828125, "grad_norm": 0.794370174407959, "learning_rate": 7.131159054949273e-05, "loss": 0.024097949266433716, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.02439, "step": 212, "tokens/total": 6410432, "tokens/train_per_sec_per_gpu": 33.19, "tokens/trainable": 96942} +{"epoch": 0.83203125, "grad_norm": 0.5890098214149475, "learning_rate": 7.104070433827139e-05, "loss": 0.009191717952489853, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00923, "step": 213, "tokens/total": 6440864, "tokens/train_per_sec_per_gpu": 33.19, "tokens/trainable": 97380} +{"epoch": 0.8359375, "grad_norm": 0.17729948461055756, "learning_rate": 7.076915060786705e-05, "loss": 0.002657170407474041, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00266, "step": 214, "tokens/total": 6471136, "tokens/train_per_sec_per_gpu": 36.92, "tokens/trainable": 97869} +{"epoch": 0.83984375, "grad_norm": 0.11908628791570663, "learning_rate": 7.049694065873882e-05, "loss": 0.0032271994277834892, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00323, "step": 215, "tokens/total": 6501328, "tokens/train_per_sec_per_gpu": 37.9, "tokens/trainable": 98320} +{"epoch": 0.84375, "grad_norm": 0.14903610944747925, "learning_rate": 7.022408581865382e-05, "loss": 0.0029319487512111664, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00294, "step": 216, "tokens/total": 6531680, "tokens/train_per_sec_per_gpu": 36.97, "tokens/trainable": 98793} +{"epoch": 0.84765625, "grad_norm": 0.10195992887020111, "learning_rate": 6.99505974422157e-05, "loss": 0.0022444254718720913, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00225, "step": 217, "tokens/total": 6562032, "tokens/train_per_sec_per_gpu": 37.33, "tokens/trainable": 99278} +{"epoch": 0.8515625, "grad_norm": 0.07194698601961136, "learning_rate": 6.967648691039213e-05, "loss": 0.0016705383313819766, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00167, "step": 218, "tokens/total": 6592304, "tokens/train_per_sec_per_gpu": 37.41, "tokens/trainable": 99751} +{"epoch": 0.85546875, "grad_norm": 0.07926305383443832, "learning_rate": 6.940176563004123e-05, "loss": 0.0014279746683314443, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00143, "step": 219, "tokens/total": 6622736, "tokens/train_per_sec_per_gpu": 32.21, "tokens/trainable": 100184} +{"epoch": 0.859375, "grad_norm": 4.5476202964782715, "learning_rate": 6.912644503343682e-05, "loss": 0.028542017564177513, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.02895, "step": 220, "tokens/total": 6653168, "tokens/train_per_sec_per_gpu": 38.12, "tokens/trainable": 100656} +{"epoch": 0.86328125, "grad_norm": 0.2030973732471466, "learning_rate": 6.885053657779273e-05, "loss": 0.002416168339550495, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00242, "step": 221, "tokens/total": 6683328, "tokens/train_per_sec_per_gpu": 32.12, "tokens/trainable": 101107} +{"epoch": 0.8671875, "grad_norm": 0.12359625846147537, "learning_rate": 6.857405174478604e-05, "loss": 0.001577290939167142, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00158, "step": 222, "tokens/total": 6713840, "tokens/train_per_sec_per_gpu": 38.57, "tokens/trainable": 101574} +{"epoch": 0.87109375, "grad_norm": 0.6420896053314209, "learning_rate": 6.82970020400792e-05, "loss": 0.012318034656345844, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.01239, "step": 223, "tokens/total": 6744224, "tokens/train_per_sec_per_gpu": 31.99, "tokens/trainable": 102015} +{"epoch": 0.875, "grad_norm": 0.35973188281059265, "learning_rate": 6.801939899284132e-05, "loss": 0.002469731029123068, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00247, "step": 224, "tokens/total": 6774432, "tokens/train_per_sec_per_gpu": 36.04, "tokens/trainable": 102463} +{"epoch": 0.87890625, "grad_norm": 0.10223028063774109, "learning_rate": 6.774125415526827e-05, "loss": 0.001325129996985197, "memory/device_reserved (GiB)": 35.94, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00133, "step": 225, "tokens/total": 6804688, "tokens/train_per_sec_per_gpu": 29.58, "tokens/trainable": 102880} +{"epoch": 0.8828125, "grad_norm": 0.7656016945838928, "learning_rate": 6.746257910210214e-05, "loss": 0.01600477285683155, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.01613, "step": 226, "tokens/total": 6835376, "tokens/train_per_sec_per_gpu": 37.94, "tokens/trainable": 103387} +{"epoch": 0.88671875, "grad_norm": 0.2979660928249359, "learning_rate": 6.718338543014937e-05, "loss": 0.004661541897803545, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00467, "step": 227, "tokens/total": 6865392, "tokens/train_per_sec_per_gpu": 40.16, "tokens/trainable": 103880} +{"epoch": 0.890625, "grad_norm": 0.4353335499763489, "learning_rate": 6.69036847577983e-05, "loss": 0.010676136240363121, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.01073, "step": 228, "tokens/total": 6895776, "tokens/train_per_sec_per_gpu": 36.82, "tokens/trainable": 104343} +{"epoch": 0.89453125, "grad_norm": 0.05993308871984482, "learning_rate": 6.662348872453553e-05, "loss": 0.0013203206472098827, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.69, "memory/max_allocated (GiB)": 33.69, "ppl": 1.00132, "step": 229, "tokens/total": 6925776, "tokens/train_per_sec_per_gpu": 33.97, "tokens/trainable": 104808} +{"epoch": 0.8984375, "grad_norm": 0.21397961676120758, "learning_rate": 6.63428089904618e-05, "loss": 0.00435336958616972, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00436, "step": 230, "tokens/total": 6956048, "tokens/train_per_sec_per_gpu": 37.4, "tokens/trainable": 105276} +{"epoch": 0.90234375, "grad_norm": 0.23647183179855347, "learning_rate": 6.60616572358065e-05, "loss": 0.007145303301513195, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00717, "step": 231, "tokens/total": 6986384, "tokens/train_per_sec_per_gpu": 34.34, "tokens/trainable": 105740} +{"epoch": 0.90625, "grad_norm": 0.01726609468460083, "learning_rate": 6.578004516044172e-05, "loss": 0.00030698676710017025, "memory/device_reserved (GiB)": 35.77, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00031, "step": 232, "tokens/total": 7016784, "tokens/train_per_sec_per_gpu": 33.52, "tokens/trainable": 106216} +{"epoch": 0.91015625, "grad_norm": 0.028660962358117104, "learning_rate": 6.549798448339548e-05, "loss": 0.000520392379257828, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00052, "step": 233, "tokens/total": 7047184, "tokens/train_per_sec_per_gpu": 38.58, "tokens/trainable": 106721} +{"epoch": 0.9140625, "grad_norm": 0.1023792177438736, "learning_rate": 6.521548694236384e-05, "loss": 0.0018057681154459715, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00181, "step": 234, "tokens/total": 7077168, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 107159} +{"epoch": 0.91796875, "grad_norm": 0.1957523226737976, "learning_rate": 6.493256429322259e-05, "loss": 0.014335989020764828, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.01444, "step": 235, "tokens/total": 7107616, "tokens/train_per_sec_per_gpu": 33.21, "tokens/trainable": 107599} +{"epoch": 0.921875, "grad_norm": 0.21969880163669586, "learning_rate": 6.464922830953799e-05, "loss": 0.00772607559338212, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00776, "step": 236, "tokens/total": 7137728, "tokens/train_per_sec_per_gpu": 34.61, "tokens/trainable": 108063} +{"epoch": 0.92578125, "grad_norm": 0.04340682178735733, "learning_rate": 6.436549078207688e-05, "loss": 0.0009402063442394137, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00094, "step": 237, "tokens/total": 7167904, "tokens/train_per_sec_per_gpu": 33.83, "tokens/trainable": 108491} +{"epoch": 0.9296875, "grad_norm": 0.31009772419929504, "learning_rate": 6.408136351831592e-05, "loss": 0.006564700044691563, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00659, "step": 238, "tokens/total": 7198288, "tokens/train_per_sec_per_gpu": 29.57, "tokens/trainable": 108919} +{"epoch": 0.93359375, "grad_norm": 0.6902568340301514, "learning_rate": 6.379685834195036e-05, "loss": 0.03366249054670334, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.03424, "step": 239, "tokens/total": 7228624, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 109360} +{"epoch": 0.9375, "grad_norm": 0.07222271710634232, "learning_rate": 6.351198709240186e-05, "loss": 0.0014452305622398853, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00145, "step": 240, "tokens/total": 7259056, "tokens/train_per_sec_per_gpu": 32.69, "tokens/trainable": 109823} +{"epoch": 0.94140625, "grad_norm": 0.19805769622325897, "learning_rate": 6.32267616243259e-05, "loss": 0.004479828290641308, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00449, "step": 241, "tokens/total": 7289328, "tokens/train_per_sec_per_gpu": 38.97, "tokens/trainable": 110281} +{"epoch": 0.9453125, "grad_norm": 0.15675081312656403, "learning_rate": 6.294119380711849e-05, "loss": 0.003017749171704054, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00302, "step": 242, "tokens/total": 7319632, "tokens/train_per_sec_per_gpu": 28.46, "tokens/trainable": 110694} +{"epoch": 0.94921875, "grad_norm": 0.024338258430361748, "learning_rate": 6.265529552442209e-05, "loss": 0.000581439642701298, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00058, "step": 243, "tokens/total": 7349936, "tokens/train_per_sec_per_gpu": 39.78, "tokens/trainable": 111188} +{"epoch": 0.953125, "grad_norm": 0.6708511114120483, "learning_rate": 6.236907867363127e-05, "loss": 0.011987617239356041, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.01206, "step": 244, "tokens/total": 7380576, "tokens/train_per_sec_per_gpu": 33.59, "tokens/trainable": 111645} +{"epoch": 0.95703125, "grad_norm": 0.05413031578063965, "learning_rate": 6.208255516539749e-05, "loss": 0.001512192189693451, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00151, "step": 245, "tokens/total": 7411200, "tokens/train_per_sec_per_gpu": 34.68, "tokens/trainable": 112128} +{"epoch": 0.9609375, "grad_norm": 0.2185056358575821, "learning_rate": 6.179573692313344e-05, "loss": 0.002352418377995491, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00236, "step": 246, "tokens/total": 7441776, "tokens/train_per_sec_per_gpu": 33.6, "tokens/trainable": 112602} +{"epoch": 0.96484375, "grad_norm": 0.057806000113487244, "learning_rate": 6.150863588251694e-05, "loss": 0.0013613419141620398, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00136, "step": 247, "tokens/total": 7470096, "tokens/train_per_sec_per_gpu": 38.27, "tokens/trainable": 113056} +{"epoch": 0.96875, "grad_norm": 0.39721816778182983, "learning_rate": 6.122126399099419e-05, "loss": 0.005575620103627443, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00559, "step": 248, "tokens/total": 7500688, "tokens/train_per_sec_per_gpu": 35.76, "tokens/trainable": 113497} +{"epoch": 0.97265625, "grad_norm": 0.24661272764205933, "learning_rate": 6.0933633207282615e-05, "loss": 0.006032303906977177, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00605, "step": 249, "tokens/total": 7530832, "tokens/train_per_sec_per_gpu": 33.19, "tokens/trainable": 113941} +{"epoch": 0.9765625, "grad_norm": 0.0430242083966732, "learning_rate": 6.064575550087316e-05, "loss": 0.001341596245765686, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00134, "step": 250, "tokens/total": 7559136, "tokens/train_per_sec_per_gpu": 34.58, "tokens/trainable": 114373} +{"epoch": 0.98046875, "grad_norm": 1.1043962240219116, "learning_rate": 6.0357642851532245e-05, "loss": 0.008739812299609184, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00878, "step": 251, "tokens/total": 7589648, "tokens/train_per_sec_per_gpu": 36.97, "tokens/trainable": 114863} +{"epoch": 0.984375, "grad_norm": 0.05025576800107956, "learning_rate": 6.0069307248803294e-05, "loss": 0.0008687145891599357, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00087, "step": 252, "tokens/total": 7619904, "tokens/train_per_sec_per_gpu": 35.74, "tokens/trainable": 115311} +{"epoch": 0.98828125, "grad_norm": 0.0689782053232193, "learning_rate": 5.9780760691507635e-05, "loss": 0.0014057592488825321, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00141, "step": 253, "tokens/total": 7650480, "tokens/train_per_sec_per_gpu": 32.83, "tokens/trainable": 115732} +{"epoch": 0.9921875, "grad_norm": 0.5817311406135559, "learning_rate": 5.9492015187245334e-05, "loss": 0.00938393920660019, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00943, "step": 254, "tokens/total": 7680912, "tokens/train_per_sec_per_gpu": 29.88, "tokens/trainable": 116164} +{"epoch": 0.99609375, "grad_norm": 0.019027426838874817, "learning_rate": 5.920308275189541e-05, "loss": 0.000433115113992244, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00043, "step": 255, "tokens/total": 7711344, "tokens/train_per_sec_per_gpu": 37.43, "tokens/trainable": 116614} +{"epoch": 1.0, "grad_norm": 0.05099409446120262, "learning_rate": 5.8913975409115874e-05, "loss": 0.0004526513221208006, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00045, "step": 256, "tokens/total": 7741936, "tokens/train_per_sec_per_gpu": 32.59, "tokens/trainable": 117062} +{"epoch": 1.00390625, "grad_norm": 0.09221355617046356, "learning_rate": 5.8624705189843395e-05, "loss": 0.0010331927333027124, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00103, "step": 257, "tokens/total": 7772272, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 117518} +{"epoch": 1.0078125, "grad_norm": 0.037981387227773666, "learning_rate": 5.833528413179249e-05, "loss": 0.0006237998022697866, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00062, "step": 258, "tokens/total": 7802560, "tokens/train_per_sec_per_gpu": 30.99, "tokens/trainable": 117948} +{"epoch": 1.01171875, "grad_norm": 0.15115460753440857, "learning_rate": 5.80457242789548e-05, "loss": 0.0022220034152269363, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00222, "step": 259, "tokens/total": 7832896, "tokens/train_per_sec_per_gpu": 32.2, "tokens/trainable": 118388} +{"epoch": 1.015625, "grad_norm": 0.03709586337208748, "learning_rate": 5.77560376810977e-05, "loss": 0.0004677239339798689, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00047, "step": 260, "tokens/total": 7863328, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 118831} +{"epoch": 1.01953125, "grad_norm": 0.05125842243432999, "learning_rate": 5.7466236393263005e-05, "loss": 0.0011223317123949528, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00112, "step": 261, "tokens/total": 7893792, "tokens/train_per_sec_per_gpu": 34.44, "tokens/trainable": 119278} +{"epoch": 1.0234375, "grad_norm": 0.2065403312444687, "learning_rate": 5.717633247526522e-05, "loss": 0.0018837190000340343, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00189, "step": 262, "tokens/total": 7924272, "tokens/train_per_sec_per_gpu": 35.36, "tokens/trainable": 119787} +{"epoch": 1.02734375, "grad_norm": 0.14627555012702942, "learning_rate": 5.688633799118971e-05, "loss": 0.0023111533373594284, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00231, "step": 263, "tokens/total": 7954624, "tokens/train_per_sec_per_gpu": 34.44, "tokens/trainable": 120276} +{"epoch": 1.03125, "grad_norm": 0.045769475400447845, "learning_rate": 5.659626500889066e-05, "loss": 0.00028873005066998303, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00029, "step": 264, "tokens/total": 7985184, "tokens/train_per_sec_per_gpu": 37.7, "tokens/trainable": 120753} +{"epoch": 1.03515625, "grad_norm": 0.3239908814430237, "learning_rate": 5.6306125599488905e-05, "loss": 0.003724604845046997, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00373, "step": 265, "tokens/total": 8015520, "tokens/train_per_sec_per_gpu": 40.07, "tokens/trainable": 121219} +{"epoch": 1.0390625, "grad_norm": 0.6028871536254883, "learning_rate": 5.601593183686955e-05, "loss": 0.01656029187142849, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.0167, "step": 266, "tokens/total": 8046064, "tokens/train_per_sec_per_gpu": 33.08, "tokens/trainable": 121694} +{"epoch": 1.04296875, "grad_norm": 0.08973632007837296, "learning_rate": 5.572569579717961e-05, "loss": 0.0009177342290058732, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00092, "step": 267, "tokens/total": 8076352, "tokens/train_per_sec_per_gpu": 40.77, "tokens/trainable": 122198} +{"epoch": 1.046875, "grad_norm": 0.007786457426846027, "learning_rate": 5.543542955832538e-05, "loss": 8.95903940545395e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00009, "step": 268, "tokens/total": 8106608, "tokens/train_per_sec_per_gpu": 35.98, "tokens/trainable": 122660} +{"epoch": 1.05078125, "grad_norm": 0.3484652042388916, "learning_rate": 5.514514519946986e-05, "loss": 0.0015085680643096566, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00151, "step": 269, "tokens/total": 8136768, "tokens/train_per_sec_per_gpu": 33.61, "tokens/trainable": 123120} +{"epoch": 1.0546875, "grad_norm": 0.005140791181474924, "learning_rate": 5.485485480053015e-05, "loss": 8.239979069912806e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00008, "step": 270, "tokens/total": 8167248, "tokens/train_per_sec_per_gpu": 36.79, "tokens/trainable": 123610} +{"epoch": 1.05859375, "grad_norm": 0.004375405143946409, "learning_rate": 5.4564570441674645e-05, "loss": 8.648644870845601e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00009, "step": 271, "tokens/total": 8197360, "tokens/train_per_sec_per_gpu": 38.27, "tokens/trainable": 124092} +{"epoch": 1.0625, "grad_norm": 0.4465174973011017, "learning_rate": 5.42743042028204e-05, "loss": 0.004284500610083342, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00429, "step": 272, "tokens/total": 8227648, "tokens/train_per_sec_per_gpu": 35.26, "tokens/trainable": 124562} +{"epoch": 1.06640625, "grad_norm": 0.15294252336025238, "learning_rate": 5.3984068163130464e-05, "loss": 0.0017521766712889075, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00175, "step": 273, "tokens/total": 8257696, "tokens/train_per_sec_per_gpu": 34.55, "tokens/trainable": 124992} +{"epoch": 1.0703125, "grad_norm": 0.26586484909057617, "learning_rate": 5.369387440051111e-05, "loss": 0.0037963378708809614, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.0038, "step": 274, "tokens/total": 8288272, "tokens/train_per_sec_per_gpu": 35.11, "tokens/trainable": 125453} +{"epoch": 1.07421875, "grad_norm": 0.0579276978969574, "learning_rate": 5.340373499110935e-05, "loss": 0.0005296460585668683, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00053, "step": 275, "tokens/total": 8318912, "tokens/train_per_sec_per_gpu": 32.01, "tokens/trainable": 125904} +{"epoch": 1.078125, "grad_norm": 0.12190263718366623, "learning_rate": 5.3113662008810304e-05, "loss": 0.0012181613128632307, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00122, "step": 276, "tokens/total": 8349280, "tokens/train_per_sec_per_gpu": 33.38, "tokens/trainable": 126378} +{"epoch": 1.08203125, "grad_norm": 0.06377701461315155, "learning_rate": 5.282366752473479e-05, "loss": 0.0005520237027667463, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00055, "step": 277, "tokens/total": 8379744, "tokens/train_per_sec_per_gpu": 34.54, "tokens/trainable": 126843} +{"epoch": 1.0859375, "grad_norm": 0.5330214500427246, "learning_rate": 5.2533763606737005e-05, "loss": 0.010966995730996132, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.01103, "step": 278, "tokens/total": 8409968, "tokens/train_per_sec_per_gpu": 38.83, "tokens/trainable": 127346} +{"epoch": 1.08984375, "grad_norm": 0.28640905022621155, "learning_rate": 5.224396231890232e-05, "loss": 0.017954371869564056, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.01812, "step": 279, "tokens/total": 8440112, "tokens/train_per_sec_per_gpu": 33.57, "tokens/trainable": 127804} +{"epoch": 1.09375, "grad_norm": 0.003401878522709012, "learning_rate": 5.195427572104522e-05, "loss": 5.5312390031758696e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00006, "step": 280, "tokens/total": 8470480, "tokens/train_per_sec_per_gpu": 33.36, "tokens/trainable": 128253} +{"epoch": 1.09765625, "grad_norm": 0.20829863846302032, "learning_rate": 5.166471586820751e-05, "loss": 0.0049300119280815125, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00494, "step": 281, "tokens/total": 8500752, "tokens/train_per_sec_per_gpu": 36.8, "tokens/trainable": 128735} +{"epoch": 1.1015625, "grad_norm": 0.010676818899810314, "learning_rate": 5.1375294810156615e-05, "loss": 0.00017420266522094607, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00017, "step": 282, "tokens/total": 8530944, "tokens/train_per_sec_per_gpu": 35.65, "tokens/trainable": 129209} +{"epoch": 1.10546875, "grad_norm": 0.1060071736574173, "learning_rate": 5.1086024590884144e-05, "loss": 0.0014481758698821068, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00145, "step": 283, "tokens/total": 8561312, "tokens/train_per_sec_per_gpu": 33.5, "tokens/trainable": 129659} +{"epoch": 1.109375, "grad_norm": 0.010597283020615578, "learning_rate": 5.079691724810461e-05, "loss": 0.00021654966985806823, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00022, "step": 284, "tokens/total": 8591712, "tokens/train_per_sec_per_gpu": 33.14, "tokens/trainable": 130106} +{"epoch": 1.11328125, "grad_norm": 0.3154262900352478, "learning_rate": 5.0507984812754684e-05, "loss": 0.00555079523473978, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00557, "step": 285, "tokens/total": 8622096, "tokens/train_per_sec_per_gpu": 31.99, "tokens/trainable": 130556} +{"epoch": 1.1171875, "grad_norm": 0.009663589298725128, "learning_rate": 5.021923930849237e-05, "loss": 0.00018602647469379008, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00019, "step": 286, "tokens/total": 8652304, "tokens/train_per_sec_per_gpu": 34.84, "tokens/trainable": 131013} +{"epoch": 1.12109375, "grad_norm": 0.02170551009476185, "learning_rate": 4.99306927511967e-05, "loss": 0.00036282287328504026, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00036, "step": 287, "tokens/total": 8682704, "tokens/train_per_sec_per_gpu": 34.57, "tokens/trainable": 131469} +{"epoch": 1.125, "grad_norm": 0.1057235449552536, "learning_rate": 4.964235714846775e-05, "loss": 0.0015502857277169824, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00155, "step": 288, "tokens/total": 8713056, "tokens/train_per_sec_per_gpu": 36.55, "tokens/trainable": 131943} +{"epoch": 1.12890625, "grad_norm": 0.09496571868658066, "learning_rate": 4.9354244499126866e-05, "loss": 0.0017094009090214968, "memory/device_reserved (GiB)": 35.91, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00171, "step": 289, "tokens/total": 8743232, "tokens/train_per_sec_per_gpu": 33.71, "tokens/trainable": 132417} +{"epoch": 1.1328125, "grad_norm": 0.023917539045214653, "learning_rate": 4.90663667927174e-05, "loss": 0.0005100768757984042, "memory/device_reserved (GiB)": 35.56, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00051, "step": 290, "tokens/total": 8773664, "tokens/train_per_sec_per_gpu": 36.96, "tokens/trainable": 132866} +{"epoch": 1.13671875, "grad_norm": 0.008018841035664082, "learning_rate": 4.877873600900581e-05, "loss": 0.00021408281463664025, "memory/device_reserved (GiB)": 35.56, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00021, "step": 291, "tokens/total": 8803920, "tokens/train_per_sec_per_gpu": 36.26, "tokens/trainable": 133367} +{"epoch": 1.140625, "grad_norm": 0.10013210028409958, "learning_rate": 4.849136411748306e-05, "loss": 0.0012666326947510242, "memory/device_reserved (GiB)": 35.58, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00127, "step": 292, "tokens/total": 8834240, "tokens/train_per_sec_per_gpu": 34.8, "tokens/trainable": 133799} +{"epoch": 1.14453125, "grad_norm": 0.04551267251372337, "learning_rate": 4.8204263076866574e-05, "loss": 0.0007189091411419213, "memory/device_reserved (GiB)": 35.58, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00072, "step": 293, "tokens/total": 8864560, "tokens/train_per_sec_per_gpu": 36.86, "tokens/trainable": 134271} +{"epoch": 1.1484375, "grad_norm": 0.028762396425008774, "learning_rate": 4.791744483460251e-05, "loss": 0.0005527742905542254, "memory/device_reserved (GiB)": 35.58, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00055, "step": 294, "tokens/total": 8895312, "tokens/train_per_sec_per_gpu": 34.05, "tokens/trainable": 134695} +{"epoch": 1.15234375, "grad_norm": 0.0814395472407341, "learning_rate": 4.7630921326368736e-05, "loss": 0.0011326507665216923, "memory/device_reserved (GiB)": 35.58, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00113, "step": 295, "tokens/total": 8925744, "tokens/train_per_sec_per_gpu": 36.1, "tokens/trainable": 135177} +{"epoch": 1.15625, "grad_norm": 0.08422354608774185, "learning_rate": 4.7344704475577916e-05, "loss": 0.0011638643918558955, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00116, "step": 296, "tokens/total": 8956048, "tokens/train_per_sec_per_gpu": 33.71, "tokens/trainable": 135611} +{"epoch": 1.16015625, "grad_norm": 0.019762732088565826, "learning_rate": 4.705880619288153e-05, "loss": 0.00038686307379975915, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00039, "step": 297, "tokens/total": 8986320, "tokens/train_per_sec_per_gpu": 33.09, "tokens/trainable": 136050} +{"epoch": 1.1640625, "grad_norm": 0.019799284636974335, "learning_rate": 4.677323837567412e-05, "loss": 0.0003641210787463933, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00036, "step": 298, "tokens/total": 9016800, "tokens/train_per_sec_per_gpu": 33.38, "tokens/trainable": 136507} +{"epoch": 1.16796875, "grad_norm": 0.010453739203512669, "learning_rate": 4.6488012907598146e-05, "loss": 0.00018282064411323518, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00018, "step": 299, "tokens/total": 9047184, "tokens/train_per_sec_per_gpu": 35.87, "tokens/trainable": 136965} +{"epoch": 1.171875, "grad_norm": 0.13788308203220367, "learning_rate": 4.620314165804964e-05, "loss": 0.002034355653449893, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00204, "step": 300, "tokens/total": 9077488, "tokens/train_per_sec_per_gpu": 34.78, "tokens/trainable": 137415} +{"epoch": 1.17578125, "grad_norm": 0.056629061698913574, "learning_rate": 4.591863648168407e-05, "loss": 0.0006158994510769844, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00062, "step": 301, "tokens/total": 9107968, "tokens/train_per_sec_per_gpu": 36.22, "tokens/trainable": 137910} +{"epoch": 1.1796875, "grad_norm": 0.027086833491921425, "learning_rate": 4.5634509217923135e-05, "loss": 0.0004739577416330576, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00047, "step": 302, "tokens/total": 9138240, "tokens/train_per_sec_per_gpu": 34.8, "tokens/trainable": 138386} +{"epoch": 1.18359375, "grad_norm": 0.035595983266830444, "learning_rate": 4.535077169046201e-05, "loss": 0.00047377185546793044, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00047, "step": 303, "tokens/total": 9168640, "tokens/train_per_sec_per_gpu": 36.96, "tokens/trainable": 138845} +{"epoch": 1.1875, "grad_norm": 0.07773973047733307, "learning_rate": 4.506743570677743e-05, "loss": 0.0008522339048795402, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00085, "step": 304, "tokens/total": 9198912, "tokens/train_per_sec_per_gpu": 31.07, "tokens/trainable": 139270} +{"epoch": 1.19140625, "grad_norm": 0.2708052098751068, "learning_rate": 4.478451305763618e-05, "loss": 0.017264865338802338, "memory/device_reserved (GiB)": 35.66, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.01741, "step": 305, "tokens/total": 9229088, "tokens/train_per_sec_per_gpu": 33.55, "tokens/trainable": 139728} +{"epoch": 1.1953125, "grad_norm": 0.10748651623725891, "learning_rate": 4.450201551660454e-05, "loss": 0.0018072875682264566, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00181, "step": 306, "tokens/total": 9259680, "tokens/train_per_sec_per_gpu": 37.68, "tokens/trainable": 140219} +{"epoch": 1.19921875, "grad_norm": 0.0025890350807458162, "learning_rate": 4.4219954839558276e-05, "loss": 4.9693619075696915e-05, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.19, "memory/max_allocated (GiB)": 33.19, "ppl": 1.00005, "step": 307, "tokens/total": 9287584, "tokens/train_per_sec_per_gpu": 34.51, "tokens/trainable": 140657} +{"epoch": 1.203125, "grad_norm": 0.018638836219906807, "learning_rate": 4.393834276419352e-05, "loss": 0.0003076127031818032, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00031, "step": 308, "tokens/total": 9318096, "tokens/train_per_sec_per_gpu": 30.38, "tokens/trainable": 141108} +{"epoch": 1.20703125, "grad_norm": 0.010553287342190742, "learning_rate": 4.36571910095382e-05, "loss": 0.00020812047296203673, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00021, "step": 309, "tokens/total": 9348544, "tokens/train_per_sec_per_gpu": 29.44, "tokens/trainable": 141531} +{"epoch": 1.2109375, "grad_norm": 0.2629532217979431, "learning_rate": 4.337651127546448e-05, "loss": 0.023792965337634087, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.02408, "step": 310, "tokens/total": 9378848, "tokens/train_per_sec_per_gpu": 35.94, "tokens/trainable": 141965} +{"epoch": 1.21484375, "grad_norm": 0.01139355544000864, "learning_rate": 4.3096315242201736e-05, "loss": 0.00018112164980266243, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00018, "step": 311, "tokens/total": 9409104, "tokens/train_per_sec_per_gpu": 34.52, "tokens/trainable": 142429} +{"epoch": 1.21875, "grad_norm": 0.01917850784957409, "learning_rate": 4.2816614569850635e-05, "loss": 0.00016473264258820564, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00016, "step": 312, "tokens/total": 9439264, "tokens/train_per_sec_per_gpu": 38.76, "tokens/trainable": 142895} +{"epoch": 1.22265625, "grad_norm": 0.4227152466773987, "learning_rate": 4.2537420897897864e-05, "loss": 0.004962727427482605, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00498, "step": 313, "tokens/total": 9469648, "tokens/train_per_sec_per_gpu": 37.12, "tokens/trainable": 143370} +{"epoch": 1.2265625, "grad_norm": 0.09607069194316864, "learning_rate": 4.225874584473174e-05, "loss": 0.0009555588476359844, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00096, "step": 314, "tokens/total": 9499968, "tokens/train_per_sec_per_gpu": 33.41, "tokens/trainable": 143826} +{"epoch": 1.23046875, "grad_norm": 0.0108202388510108, "learning_rate": 4.19806010071587e-05, "loss": 0.0002800720394589007, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.00028, "step": 315, "tokens/total": 9530448, "tokens/train_per_sec_per_gpu": 35.11, "tokens/trainable": 144308} +{"epoch": 1.234375, "grad_norm": 0.5568331480026245, "learning_rate": 4.170299795992081e-05, "loss": 0.0007924925885163248, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00079, "step": 316, "tokens/total": 9560848, "tokens/train_per_sec_per_gpu": 36.0, "tokens/trainable": 144765} +{"epoch": 1.23828125, "grad_norm": 0.3749312162399292, "learning_rate": 4.142594825521398e-05, "loss": 0.009996296837925911, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01005, "step": 317, "tokens/total": 9591376, "tokens/train_per_sec_per_gpu": 37.75, "tokens/trainable": 145265} +{"epoch": 1.2421875, "grad_norm": 0.22495749592781067, "learning_rate": 4.114946342220728e-05, "loss": 0.002975808223709464, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00298, "step": 318, "tokens/total": 9621728, "tokens/train_per_sec_per_gpu": 37.25, "tokens/trainable": 145724} +{"epoch": 1.24609375, "grad_norm": 0.47026923298835754, "learning_rate": 4.087355496656321e-05, "loss": 0.006408916786313057, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00643, "step": 319, "tokens/total": 9652272, "tokens/train_per_sec_per_gpu": 37.72, "tokens/trainable": 146219} +{"epoch": 1.25, "grad_norm": 0.17889857292175293, "learning_rate": 4.05982343699588e-05, "loss": 0.0023502246476709843, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00235, "step": 320, "tokens/total": 9682672, "tokens/train_per_sec_per_gpu": 34.58, "tokens/trainable": 146693} +{"epoch": 1.25390625, "grad_norm": 0.062361083924770355, "learning_rate": 4.0323513089607876e-05, "loss": 0.0006638169870711863, "memory/device_reserved (GiB)": 35.9, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00066, "step": 321, "tokens/total": 9712832, "tokens/train_per_sec_per_gpu": 35.87, "tokens/trainable": 147139} +{"epoch": 1.2578125, "grad_norm": 0.009022507816553116, "learning_rate": 4.004940255778431e-05, "loss": 0.00017464160919189453, "memory/device_reserved (GiB)": 36.14, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00017, "step": 322, "tokens/total": 9743344, "tokens/train_per_sec_per_gpu": 35.74, "tokens/trainable": 147589} +{"epoch": 1.26171875, "grad_norm": 0.38307738304138184, "learning_rate": 3.977591418134619e-05, "loss": 0.0034184439573436975, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00342, "step": 323, "tokens/total": 9773760, "tokens/train_per_sec_per_gpu": 35.77, "tokens/trainable": 148045} +{"epoch": 1.265625, "grad_norm": 0.3203769028186798, "learning_rate": 3.95030593412612e-05, "loss": 0.0051149846985936165, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.01, "memory/max_allocated (GiB)": 34.01, "ppl": 1.00513, "step": 324, "tokens/total": 9804080, "tokens/train_per_sec_per_gpu": 37.47, "tokens/trainable": 148494} +{"epoch": 1.26953125, "grad_norm": 0.7488558292388916, "learning_rate": 3.923084939213296e-05, "loss": 0.01595677062869072, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.01608, "step": 325, "tokens/total": 9834400, "tokens/train_per_sec_per_gpu": 36.84, "tokens/trainable": 148957} +{"epoch": 1.2734375, "grad_norm": 0.009091824293136597, "learning_rate": 3.895929566172861e-05, "loss": 0.00014956737868487835, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00015, "step": 326, "tokens/total": 9864704, "tokens/train_per_sec_per_gpu": 38.41, "tokens/trainable": 149451} +{"epoch": 1.27734375, "grad_norm": 0.005110942758619785, "learning_rate": 3.868840945050728e-05, "loss": 0.00013913170550949872, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00014, "step": 327, "tokens/total": 9894656, "tokens/train_per_sec_per_gpu": 40.07, "tokens/trainable": 149904} +{"epoch": 1.28125, "grad_norm": 0.007466079201549292, "learning_rate": 3.841820203114995e-05, "loss": 0.00013769841461908072, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00014, "step": 328, "tokens/total": 9924976, "tokens/train_per_sec_per_gpu": 34.36, "tokens/trainable": 150378} +{"epoch": 1.28515625, "grad_norm": 0.16560006141662598, "learning_rate": 3.814868464809027e-05, "loss": 0.009101686999201775, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00914, "step": 329, "tokens/total": 9955376, "tokens/train_per_sec_per_gpu": 32.32, "tokens/trainable": 150853} +{"epoch": 1.2890625, "grad_norm": 0.01060777809470892, "learning_rate": 3.787986851704667e-05, "loss": 0.00019600678933784366, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.0002, "step": 330, "tokens/total": 9985696, "tokens/train_per_sec_per_gpu": 34.45, "tokens/trainable": 151289} +{"epoch": 1.29296875, "grad_norm": 0.0109481830149889, "learning_rate": 3.7611764824555654e-05, "loss": 0.00029803148936480284, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.0003, "step": 331, "tokens/total": 10016112, "tokens/train_per_sec_per_gpu": 37.68, "tokens/trainable": 151757} +{"epoch": 1.296875, "grad_norm": 0.011753400787711143, "learning_rate": 3.734438472750619e-05, "loss": 0.00021420676785055548, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00021, "step": 332, "tokens/total": 10046544, "tokens/train_per_sec_per_gpu": 34.55, "tokens/trainable": 152207} +{"epoch": 1.30078125, "grad_norm": 0.3032369613647461, "learning_rate": 3.707773935267552e-05, "loss": 0.004725611303001642, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00474, "step": 333, "tokens/total": 10077120, "tokens/train_per_sec_per_gpu": 32.34, "tokens/trainable": 152646} +{"epoch": 1.3046875, "grad_norm": 0.024216441437602043, "learning_rate": 3.68118397962661e-05, "loss": 0.0005056065274402499, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00051, "step": 334, "tokens/total": 10107600, "tokens/train_per_sec_per_gpu": 33.93, "tokens/trainable": 153063} +{"epoch": 1.30859375, "grad_norm": 0.10152052342891693, "learning_rate": 3.654669712344384e-05, "loss": 0.001724423374980688, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00173, "step": 335, "tokens/total": 10137712, "tokens/train_per_sec_per_gpu": 30.92, "tokens/trainable": 153493} +{"epoch": 1.3125, "grad_norm": 0.06117634102702141, "learning_rate": 3.628232236787763e-05, "loss": 0.0013021706836298108, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.0013, "step": 336, "tokens/total": 10168096, "tokens/train_per_sec_per_gpu": 33.65, "tokens/trainable": 153945} +{"epoch": 1.31640625, "grad_norm": 0.022510197013616562, "learning_rate": 3.6018726531280144e-05, "loss": 0.0003054860280826688, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00031, "step": 337, "tokens/total": 10198720, "tokens/train_per_sec_per_gpu": 35.4, "tokens/trainable": 154432} +{"epoch": 1.3203125, "grad_norm": 0.19209441542625427, "learning_rate": 3.575592058295017e-05, "loss": 0.0020988616161048412, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.0021, "step": 338, "tokens/total": 10228976, "tokens/train_per_sec_per_gpu": 34.47, "tokens/trainable": 154881} +{"epoch": 1.32421875, "grad_norm": 0.008805932477116585, "learning_rate": 3.549391545931585e-05, "loss": 0.00020574698282871395, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00021, "step": 339, "tokens/total": 10259440, "tokens/train_per_sec_per_gpu": 30.32, "tokens/trainable": 155306} +{"epoch": 1.328125, "grad_norm": 0.01874561607837677, "learning_rate": 3.5232722063479914e-05, "loss": 0.0003597445320338011, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00036, "step": 340, "tokens/total": 10289824, "tokens/train_per_sec_per_gpu": 31.7, "tokens/trainable": 155719} +{"epoch": 1.33203125, "grad_norm": 0.005896273069083691, "learning_rate": 3.49723512647657e-05, "loss": 0.00010345762711949646, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.0001, "step": 341, "tokens/total": 10320144, "tokens/train_per_sec_per_gpu": 36.3, "tokens/trainable": 156209} +{"epoch": 1.3359375, "grad_norm": 0.010024541057646275, "learning_rate": 3.471281389826491e-05, "loss": 0.0001570192980580032, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00016, "step": 342, "tokens/total": 10350528, "tokens/train_per_sec_per_gpu": 34.17, "tokens/trainable": 156656} +{"epoch": 1.33984375, "grad_norm": 0.10390307009220123, "learning_rate": 3.4454120764386764e-05, "loss": 0.0013503385707736015, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00135, "step": 343, "tokens/total": 10380816, "tokens/train_per_sec_per_gpu": 34.23, "tokens/trainable": 157118} +{"epoch": 1.34375, "grad_norm": 0.049824248999357224, "learning_rate": 3.4196282628408526e-05, "loss": 0.0005016025970689952, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0005, "step": 344, "tokens/total": 10411312, "tokens/train_per_sec_per_gpu": 33.56, "tokens/trainable": 157562} +{"epoch": 1.34765625, "grad_norm": 0.011971387080848217, "learning_rate": 3.3939310220027456e-05, "loss": 0.00018647368415258825, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00019, "step": 345, "tokens/total": 10441696, "tokens/train_per_sec_per_gpu": 35.7, "tokens/trainable": 158003} +{"epoch": 1.3515625, "grad_norm": 0.020012276247143745, "learning_rate": 3.3683214232914404e-05, "loss": 0.0004137884243391454, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00041, "step": 346, "tokens/total": 10472032, "tokens/train_per_sec_per_gpu": 29.4, "tokens/trainable": 158419} +{"epoch": 1.35546875, "grad_norm": 0.1378275603055954, "learning_rate": 3.342800532426873e-05, "loss": 0.001066669705323875, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00107, "step": 347, "tokens/total": 10502176, "tokens/train_per_sec_per_gpu": 36.17, "tokens/trainable": 158877} +{"epoch": 1.359375, "grad_norm": 0.012419048696756363, "learning_rate": 3.317369411437484e-05, "loss": 0.0002563716552685946, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.0, "memory/max_allocated (GiB)": 34.0, "ppl": 1.00026, "step": 348, "tokens/total": 10532896, "tokens/train_per_sec_per_gpu": 29.19, "tokens/trainable": 159299} +{"epoch": 1.36328125, "grad_norm": 0.0319787822663784, "learning_rate": 3.292029118616024e-05, "loss": 0.0005879281088709831, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00059, "step": 349, "tokens/total": 10563024, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 159750} +{"epoch": 1.3671875, "grad_norm": 0.009861056692898273, "learning_rate": 3.266780708475511e-05, "loss": 0.00010148907313123345, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.99, "memory/max_allocated (GiB)": 33.99, "ppl": 1.0001, "step": 350, "tokens/total": 10593536, "tokens/train_per_sec_per_gpu": 35.76, "tokens/trainable": 160217} +{"epoch": 1.37109375, "grad_norm": 0.005884335841983557, "learning_rate": 3.241625231705354e-05, "loss": 0.00010095423203893006, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.0001, "step": 351, "tokens/total": 10623776, "tokens/train_per_sec_per_gpu": 33.11, "tokens/trainable": 160675} +{"epoch": 1.375, "grad_norm": 0.24247600138187408, "learning_rate": 3.216563735127618e-05, "loss": 0.0036722179502248764, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00368, "step": 352, "tokens/total": 10654032, "tokens/train_per_sec_per_gpu": 35.02, "tokens/trainable": 161132} +{"epoch": 1.37890625, "grad_norm": 0.0033956863917410374, "learning_rate": 3.191597261653475e-05, "loss": 6.495713023468852e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00006, "step": 353, "tokens/total": 10684352, "tokens/train_per_sec_per_gpu": 29.4, "tokens/trainable": 161585} +{"epoch": 1.3828125, "grad_norm": 0.8877756595611572, "learning_rate": 3.166726850239794e-05, "loss": 0.008384269662201405, "memory/device_reserved (GiB)": 35.41, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00842, "step": 354, "tokens/total": 10714816, "tokens/train_per_sec_per_gpu": 37.32, "tokens/trainable": 162050} +{"epoch": 1.38671875, "grad_norm": 0.00449851481243968, "learning_rate": 3.141953535845912e-05, "loss": 8.070325566222891e-05, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00008, "step": 355, "tokens/total": 10745296, "tokens/train_per_sec_per_gpu": 32.65, "tokens/trainable": 162478} +{"epoch": 1.390625, "grad_norm": 0.0048718261532485485, "learning_rate": 3.11727834939056e-05, "loss": 7.578312943223864e-05, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.61, "memory/max_allocated (GiB)": 33.61, "ppl": 1.00008, "step": 356, "tokens/total": 10775296, "tokens/train_per_sec_per_gpu": 28.74, "tokens/trainable": 162893} +{"epoch": 1.39453125, "grad_norm": 0.21484895050525665, "learning_rate": 3.092702317708967e-05, "loss": 0.0030353569891303778, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00304, "step": 357, "tokens/total": 10805600, "tokens/train_per_sec_per_gpu": 35.49, "tokens/trainable": 163347} +{"epoch": 1.3984375, "grad_norm": 0.07261230796575546, "learning_rate": 3.0682264635101276e-05, "loss": 0.0004826942749787122, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00048, "step": 358, "tokens/total": 10835760, "tokens/train_per_sec_per_gpu": 33.58, "tokens/trainable": 163796} +{"epoch": 1.40234375, "grad_norm": 0.022041240707039833, "learning_rate": 3.0438518053342407e-05, "loss": 0.00020454936020541936, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.0002, "step": 359, "tokens/total": 10866208, "tokens/train_per_sec_per_gpu": 32.47, "tokens/trainable": 164237} +{"epoch": 1.40625, "grad_norm": 0.005708560813218355, "learning_rate": 3.0195793575103266e-05, "loss": 0.00012487702770158648, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00012, "step": 360, "tokens/total": 10896672, "tokens/train_per_sec_per_gpu": 34.69, "tokens/trainable": 164684} +{"epoch": 1.41015625, "grad_norm": 0.02250049076974392, "learning_rate": 2.9954101301140146e-05, "loss": 0.00025320789427496493, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00025, "step": 361, "tokens/total": 10926944, "tokens/train_per_sec_per_gpu": 37.05, "tokens/trainable": 165146} +{"epoch": 1.4140625, "grad_norm": 0.013911883346736431, "learning_rate": 2.9713451289255123e-05, "loss": 0.00022382009774446487, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00022, "step": 362, "tokens/total": 10957456, "tokens/train_per_sec_per_gpu": 35.37, "tokens/trainable": 165632} +{"epoch": 1.41796875, "grad_norm": 0.2468303143978119, "learning_rate": 2.9473853553877484e-05, "loss": 0.01283702440559864, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.01292, "step": 363, "tokens/total": 10987696, "tokens/train_per_sec_per_gpu": 34.66, "tokens/trainable": 166080} +{"epoch": 1.421875, "grad_norm": 0.008053003810346127, "learning_rate": 2.9235318065647e-05, "loss": 0.00013117909838911146, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00013, "step": 364, "tokens/total": 11017904, "tokens/train_per_sec_per_gpu": 34.62, "tokens/trainable": 166531} +{"epoch": 1.42578125, "grad_norm": 0.04743769019842148, "learning_rate": 2.8997854750998964e-05, "loss": 0.00024472299264743924, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00024, "step": 365, "tokens/total": 11048224, "tokens/train_per_sec_per_gpu": 36.91, "tokens/trainable": 167019} +{"epoch": 1.4296875, "grad_norm": 0.008861004374921322, "learning_rate": 2.8761473491751258e-05, "loss": 0.00010162356193177402, "memory/device_reserved (GiB)": 35.84, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.0001, "step": 366, "tokens/total": 11078544, "tokens/train_per_sec_per_gpu": 30.94, "tokens/trainable": 167469} +{"epoch": 1.43359375, "grad_norm": 0.06953191757202148, "learning_rate": 2.8526184124692883e-05, "loss": 0.0006842018919996917, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00068, "step": 367, "tokens/total": 11109088, "tokens/train_per_sec_per_gpu": 31.86, "tokens/trainable": 167914} +{"epoch": 1.4375, "grad_norm": 0.10965435951948166, "learning_rate": 2.829199644117484e-05, "loss": 0.0006139783654361963, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00061, "step": 368, "tokens/total": 11139408, "tokens/train_per_sec_per_gpu": 36.16, "tokens/trainable": 168357} +{"epoch": 1.44140625, "grad_norm": 0.013294316828250885, "learning_rate": 2.8058920186702553e-05, "loss": 0.0001493502495577559, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00015, "step": 369, "tokens/total": 11169920, "tokens/train_per_sec_per_gpu": 32.75, "tokens/trainable": 168832} +{"epoch": 1.4453125, "grad_norm": 0.7463682889938354, "learning_rate": 2.782696506053033e-05, "loss": 0.013401923701167107, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.01349, "step": 370, "tokens/total": 11200288, "tokens/train_per_sec_per_gpu": 37.41, "tokens/trainable": 169333} +{"epoch": 1.44921875, "grad_norm": 0.001093173515982926, "learning_rate": 2.7596140715257824e-05, "loss": 2.9635479222633876e-05, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00003, "step": 371, "tokens/total": 11230832, "tokens/train_per_sec_per_gpu": 35.62, "tokens/trainable": 169774} +{"epoch": 1.453125, "grad_norm": 0.1324324756860733, "learning_rate": 2.7366456756428184e-05, "loss": 0.0015135211870074272, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00151, "step": 372, "tokens/total": 11261264, "tokens/train_per_sec_per_gpu": 33.41, "tokens/trainable": 170241} +{"epoch": 1.45703125, "grad_norm": 0.006301951594650745, "learning_rate": 2.7137922742128486e-05, "loss": 0.0001302927266806364, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00013, "step": 373, "tokens/total": 11291456, "tokens/train_per_sec_per_gpu": 33.69, "tokens/trainable": 170706} +{"epoch": 1.4609375, "grad_norm": 0.011703136377036572, "learning_rate": 2.691054818259188e-05, "loss": 0.00020681106252595782, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00021, "step": 374, "tokens/total": 11321952, "tokens/train_per_sec_per_gpu": 32.53, "tokens/trainable": 171182} +{"epoch": 1.46484375, "grad_norm": 0.015446359291672707, "learning_rate": 2.6684342539801933e-05, "loss": 0.0003060699673369527, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00031, "step": 375, "tokens/total": 11352096, "tokens/train_per_sec_per_gpu": 34.75, "tokens/trainable": 171646} +{"epoch": 1.46875, "grad_norm": 0.008862881921231747, "learning_rate": 2.645931522709877e-05, "loss": 0.00019395005074329674, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00019, "step": 376, "tokens/total": 11382528, "tokens/train_per_sec_per_gpu": 28.57, "tokens/trainable": 172063} +{"epoch": 1.47265625, "grad_norm": 0.03388524800539017, "learning_rate": 2.6235475608787365e-05, "loss": 0.0004883318324573338, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00049, "step": 377, "tokens/total": 11412832, "tokens/train_per_sec_per_gpu": 36.63, "tokens/trainable": 172539} +{"epoch": 1.4765625, "grad_norm": 0.02546733431518078, "learning_rate": 2.6012832999747916e-05, "loss": 0.0005159692373126745, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00052, "step": 378, "tokens/total": 11442864, "tokens/train_per_sec_per_gpu": 34.99, "tokens/trainable": 172997} +{"epoch": 1.48046875, "grad_norm": 0.045234713703393936, "learning_rate": 2.579139666504821e-05, "loss": 0.0004851966805290431, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00049, "step": 379, "tokens/total": 11473328, "tokens/train_per_sec_per_gpu": 36.26, "tokens/trainable": 173481} +{"epoch": 1.484375, "grad_norm": 0.011559339240193367, "learning_rate": 2.557117581955798e-05, "loss": 0.00026303555932827294, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00026, "step": 380, "tokens/total": 11503504, "tokens/train_per_sec_per_gpu": 30.94, "tokens/trainable": 173946} +{"epoch": 1.48828125, "grad_norm": 0.01947280764579773, "learning_rate": 2.5352179627565532e-05, "loss": 0.0004145601997151971, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00041, "step": 381, "tokens/total": 11533920, "tokens/train_per_sec_per_gpu": 34.7, "tokens/trainable": 174417} +{"epoch": 1.4921875, "grad_norm": 0.03896784782409668, "learning_rate": 2.5134417202396277e-05, "loss": 0.0005209866212680936, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00052, "step": 382, "tokens/total": 11564128, "tokens/train_per_sec_per_gpu": 33.17, "tokens/trainable": 174872} +{"epoch": 1.49609375, "grad_norm": 0.012542990036308765, "learning_rate": 2.491789760603361e-05, "loss": 0.00026536238146945834, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.00027, "step": 383, "tokens/total": 11594672, "tokens/train_per_sec_per_gpu": 31.99, "tokens/trainable": 175313} +{"epoch": 1.5, "grad_norm": 0.23138198256492615, "learning_rate": 2.4702629848741764e-05, "loss": 0.011072758585214615, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.01113, "step": 384, "tokens/total": 11625056, "tokens/train_per_sec_per_gpu": 32.96, "tokens/trainable": 175750} +{"epoch": 1.50390625, "grad_norm": 0.5275915861129761, "learning_rate": 2.4488622888690785e-05, "loss": 0.016837235540151596, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.01698, "step": 385, "tokens/total": 11655488, "tokens/train_per_sec_per_gpu": 35.8, "tokens/trainable": 176232} +{"epoch": 1.5078125, "grad_norm": 0.008643269538879395, "learning_rate": 2.427588563158384e-05, "loss": 0.00011418825306463987, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00011, "step": 386, "tokens/total": 11685968, "tokens/train_per_sec_per_gpu": 34.9, "tokens/trainable": 176680} +{"epoch": 1.51171875, "grad_norm": 0.00684578949585557, "learning_rate": 2.406442693028651e-05, "loss": 0.0001287878112634644, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00013, "step": 387, "tokens/total": 11716176, "tokens/train_per_sec_per_gpu": 33.24, "tokens/trainable": 177144} +{"epoch": 1.515625, "grad_norm": 0.07777013629674911, "learning_rate": 2.3854255584458547e-05, "loss": 0.001054117688909173, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00105, "step": 388, "tokens/total": 11746704, "tokens/train_per_sec_per_gpu": 36.67, "tokens/trainable": 177630} +{"epoch": 1.51953125, "grad_norm": 0.021813517436385155, "learning_rate": 2.3645380340187508e-05, "loss": 0.00045925029553472996, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00046, "step": 389, "tokens/total": 11777008, "tokens/train_per_sec_per_gpu": 33.51, "tokens/trainable": 178079} +{"epoch": 1.5234375, "grad_norm": 0.2766229808330536, "learning_rate": 2.3437809889624914e-05, "loss": 0.0063839266076684, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0064, "step": 390, "tokens/total": 11807168, "tokens/train_per_sec_per_gpu": 36.99, "tokens/trainable": 178552} +{"epoch": 1.52734375, "grad_norm": 0.5076630711555481, "learning_rate": 2.3231552870624487e-05, "loss": 0.003972330130636692, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00398, "step": 391, "tokens/total": 11837776, "tokens/train_per_sec_per_gpu": 34.49, "tokens/trainable": 179021} +{"epoch": 1.53125, "grad_norm": 0.046427834779024124, "learning_rate": 2.3026617866382657e-05, "loss": 0.00048776037874631584, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00049, "step": 392, "tokens/total": 11867968, "tokens/train_per_sec_per_gpu": 33.57, "tokens/trainable": 179478} +{"epoch": 1.53515625, "grad_norm": 0.12148728966712952, "learning_rate": 2.2823013405081507e-05, "loss": 0.00316976523026824, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00317, "step": 393, "tokens/total": 11898320, "tokens/train_per_sec_per_gpu": 33.27, "tokens/trainable": 179910} +{"epoch": 1.5390625, "grad_norm": 0.02432228811085224, "learning_rate": 2.2620747959533722e-05, "loss": 0.00029762828489765525, "memory/device_reserved (GiB)": 35.87, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.0003, "step": 394, "tokens/total": 11928624, "tokens/train_per_sec_per_gpu": 34.45, "tokens/trainable": 180361} +{"epoch": 1.54296875, "grad_norm": 0.01567767933011055, "learning_rate": 2.2419829946830123e-05, "loss": 0.00036671021371148527, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00037, "step": 395, "tokens/total": 11959088, "tokens/train_per_sec_per_gpu": 32.31, "tokens/trainable": 180774} +{"epoch": 1.546875, "grad_norm": 0.02068573608994484, "learning_rate": 2.2220267727989325e-05, "loss": 0.00030358770163729787, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.0003, "step": 396, "tokens/total": 11989504, "tokens/train_per_sec_per_gpu": 35.69, "tokens/trainable": 181246} +{"epoch": 1.55078125, "grad_norm": 0.2614375948905945, "learning_rate": 2.202206960760984e-05, "loss": 0.00875169225037098, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00879, "step": 397, "tokens/total": 12019904, "tokens/train_per_sec_per_gpu": 33.65, "tokens/trainable": 181734} +{"epoch": 1.5546875, "grad_norm": 0.037434663623571396, "learning_rate": 2.182524383352446e-05, "loss": 0.00044167632586322725, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00044, "step": 398, "tokens/total": 12050384, "tokens/train_per_sec_per_gpu": 36.01, "tokens/trainable": 182169} +{"epoch": 1.55859375, "grad_norm": 0.16984692215919495, "learning_rate": 2.1629798596457056e-05, "loss": 0.0007633643108420074, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00076, "step": 399, "tokens/total": 12080576, "tokens/train_per_sec_per_gpu": 35.34, "tokens/trainable": 182641} +{"epoch": 1.5625, "grad_norm": 0.17288610339164734, "learning_rate": 2.1435742029681725e-05, "loss": 0.0036359927617013454, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.00364, "step": 400, "tokens/total": 12110896, "tokens/train_per_sec_per_gpu": 26.93, "tokens/trainable": 183056} +{"epoch": 1.56640625, "grad_norm": 0.056392524391412735, "learning_rate": 2.124308220868431e-05, "loss": 0.0008229271625168622, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.71, "memory/max_allocated (GiB)": 33.71, "ppl": 1.00082, "step": 401, "tokens/total": 12140784, "tokens/train_per_sec_per_gpu": 32.32, "tokens/trainable": 183504} +{"epoch": 1.5703125, "grad_norm": 0.0365372970700264, "learning_rate": 2.105182715082638e-05, "loss": 0.0007215660298243165, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00072, "step": 402, "tokens/total": 12171024, "tokens/train_per_sec_per_gpu": 39.29, "tokens/trainable": 184004} +{"epoch": 1.57421875, "grad_norm": 0.016568642109632492, "learning_rate": 2.0861984815011552e-05, "loss": 0.00029834112501703203, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.0003, "step": 403, "tokens/total": 12201184, "tokens/train_per_sec_per_gpu": 33.04, "tokens/trainable": 184476} +{"epoch": 1.578125, "grad_norm": 0.06592518836259842, "learning_rate": 2.0673563101354323e-05, "loss": 0.0007513194577768445, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00075, "step": 404, "tokens/total": 12231232, "tokens/train_per_sec_per_gpu": 32.35, "tokens/trainable": 184905} +{"epoch": 1.58203125, "grad_norm": 0.013345538638532162, "learning_rate": 2.0486569850851317e-05, "loss": 0.0001536268973723054, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.72, "memory/max_allocated (GiB)": 33.72, "ppl": 1.00015, "step": 405, "tokens/total": 12261280, "tokens/train_per_sec_per_gpu": 29.82, "tokens/trainable": 185338} +{"epoch": 1.5859375, "grad_norm": 0.11538074165582657, "learning_rate": 2.0301012845054956e-05, "loss": 0.001165087684057653, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00117, "step": 406, "tokens/total": 12291520, "tokens/train_per_sec_per_gpu": 32.02, "tokens/trainable": 185792} +{"epoch": 1.58984375, "grad_norm": 0.01569611392915249, "learning_rate": 2.011689980574966e-05, "loss": 0.00031081197084859014, "memory/device_reserved (GiB)": 35.88, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00031, "step": 407, "tokens/total": 12321824, "tokens/train_per_sec_per_gpu": 37.41, "tokens/trainable": 186251} +{"epoch": 1.59375, "grad_norm": 0.02253701724112034, "learning_rate": 1.993423839463052e-05, "loss": 0.00034153330489061773, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00034, "step": 408, "tokens/total": 12352176, "tokens/train_per_sec_per_gpu": 33.33, "tokens/trainable": 186689} +{"epoch": 1.59765625, "grad_norm": 0.40578708052635193, "learning_rate": 1.975303621298445e-05, "loss": 0.005957326851785183, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00598, "step": 409, "tokens/total": 12382608, "tokens/train_per_sec_per_gpu": 34.7, "tokens/trainable": 187121} +{"epoch": 1.6015625, "grad_norm": 0.15000925958156586, "learning_rate": 1.957330080137385e-05, "loss": 0.0013339307624846697, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00133, "step": 410, "tokens/total": 12413152, "tokens/train_per_sec_per_gpu": 35.55, "tokens/trainable": 187568} +{"epoch": 1.60546875, "grad_norm": 0.006941859144717455, "learning_rate": 1.9395039639322864e-05, "loss": 0.00013002814375795424, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00013, "step": 411, "tokens/total": 12443600, "tokens/train_per_sec_per_gpu": 34.76, "tokens/trainable": 188010} +{"epoch": 1.609375, "grad_norm": 0.12384258210659027, "learning_rate": 1.9218260145006073e-05, "loss": 0.0010012347484007478, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.001, "step": 412, "tokens/total": 12471952, "tokens/train_per_sec_per_gpu": 39.3, "tokens/trainable": 188446} +{"epoch": 1.61328125, "grad_norm": 0.3418160378932953, "learning_rate": 1.904296967493982e-05, "loss": 0.00740136718377471, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00743, "step": 413, "tokens/total": 12502480, "tokens/train_per_sec_per_gpu": 32.07, "tokens/trainable": 188897} +{"epoch": 1.6171875, "grad_norm": 0.006758023519068956, "learning_rate": 1.8869175523676064e-05, "loss": 6.429506174754351e-05, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00006, "step": 414, "tokens/total": 12532720, "tokens/train_per_sec_per_gpu": 37.49, "tokens/trainable": 189366} +{"epoch": 1.62109375, "grad_norm": 0.19964486360549927, "learning_rate": 1.869688492349885e-05, "loss": 0.005143567454069853, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00516, "step": 415, "tokens/total": 12562848, "tokens/train_per_sec_per_gpu": 32.4, "tokens/trainable": 189779} +{"epoch": 1.625, "grad_norm": 0.16783277690410614, "learning_rate": 1.85261050441233e-05, "loss": 0.0007222781423479319, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00072, "step": 416, "tokens/total": 12593328, "tokens/train_per_sec_per_gpu": 34.56, "tokens/trainable": 190245} +{"epoch": 1.62890625, "grad_norm": 0.018791699782013893, "learning_rate": 1.8356842992397304e-05, "loss": 0.0002913455246016383, "memory/device_reserved (GiB)": 35.96, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00029, "step": 417, "tokens/total": 12623760, "tokens/train_per_sec_per_gpu": 35.59, "tokens/trainable": 190737} +{"epoch": 1.6328125, "grad_norm": 0.09821721911430359, "learning_rate": 1.8189105812005714e-05, "loss": 0.0005217839498072863, "memory/device_reserved (GiB)": 35.81, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00052, "step": 418, "tokens/total": 12654272, "tokens/train_per_sec_per_gpu": 31.38, "tokens/trainable": 191155} +{"epoch": 1.63671875, "grad_norm": 0.071963369846344, "learning_rate": 1.802290048317732e-05, "loss": 0.0007684896700084209, "memory/device_reserved (GiB)": 35.81, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00077, "step": 419, "tokens/total": 12684544, "tokens/train_per_sec_per_gpu": 31.98, "tokens/trainable": 191577} +{"epoch": 1.640625, "grad_norm": 0.041365377604961395, "learning_rate": 1.785823392239424e-05, "loss": 0.0007164644775912166, "memory/device_reserved (GiB)": 35.81, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00072, "step": 420, "tokens/total": 12715008, "tokens/train_per_sec_per_gpu": 37.25, "tokens/trainable": 192051} +{"epoch": 1.64453125, "grad_norm": 0.01246409397572279, "learning_rate": 1.7695112982104225e-05, "loss": 0.0002488879836164415, "memory/device_reserved (GiB)": 35.81, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00025, "step": 421, "tokens/total": 12745232, "tokens/train_per_sec_per_gpu": 31.77, "tokens/trainable": 192500} +{"epoch": 1.6484375, "grad_norm": 0.00797255989164114, "learning_rate": 1.7533544450435433e-05, "loss": 8.724055078346282e-05, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00009, "step": 422, "tokens/total": 12775824, "tokens/train_per_sec_per_gpu": 33.23, "tokens/trainable": 192955} +{"epoch": 1.65234375, "grad_norm": 0.029961727559566498, "learning_rate": 1.7373535050913946e-05, "loss": 0.00036373038892634213, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00036, "step": 423, "tokens/total": 12806240, "tokens/train_per_sec_per_gpu": 38.03, "tokens/trainable": 193462} +{"epoch": 1.65625, "grad_norm": 0.11438705027103424, "learning_rate": 1.721509144218405e-05, "loss": 0.0008654020493850112, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00087, "step": 424, "tokens/total": 12836512, "tokens/train_per_sec_per_gpu": 37.0, "tokens/trainable": 193943} +{"epoch": 1.66015625, "grad_norm": 0.11104747653007507, "learning_rate": 1.705822021773101e-05, "loss": 0.0019330887589603662, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00193, "step": 425, "tokens/total": 12866704, "tokens/train_per_sec_per_gpu": 29.75, "tokens/trainable": 194360} +{"epoch": 1.6640625, "grad_norm": 0.005441099405288696, "learning_rate": 1.69029279056068e-05, "loss": 0.00012030347716063261, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00012, "step": 426, "tokens/total": 12896912, "tokens/train_per_sec_per_gpu": 35.43, "tokens/trainable": 194848} +{"epoch": 1.66796875, "grad_norm": 0.010912942700088024, "learning_rate": 1.6749220968158415e-05, "loss": 0.00012806981976609677, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00013, "step": 427, "tokens/total": 12927248, "tokens/train_per_sec_per_gpu": 34.48, "tokens/trainable": 195314} +{"epoch": 1.671875, "grad_norm": 0.007124146446585655, "learning_rate": 1.659710580175893e-05, "loss": 9.905424667522311e-05, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.0001, "step": 428, "tokens/total": 12957632, "tokens/train_per_sec_per_gpu": 32.4, "tokens/trainable": 195732} +{"epoch": 1.67578125, "grad_norm": 0.23617680370807648, "learning_rate": 1.644658873654133e-05, "loss": 0.0024689119309186935, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.75, "memory/max_allocated (GiB)": 33.75, "ppl": 1.00247, "step": 429, "tokens/total": 12987824, "tokens/train_per_sec_per_gpu": 33.37, "tokens/trainable": 196197} +{"epoch": 1.6796875, "grad_norm": 0.29176223278045654, "learning_rate": 1.629767603613508e-05, "loss": 0.005283118691295385, "memory/device_reserved (GiB)": 35.83, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.0053, "step": 430, "tokens/total": 13018208, "tokens/train_per_sec_per_gpu": 32.85, "tokens/trainable": 196678} +{"epoch": 1.68359375, "grad_norm": 0.030829036608338356, "learning_rate": 1.615037389740547e-05, "loss": 0.0004377455043140799, "memory/device_reserved (GiB)": 36.36, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.00044, "step": 431, "tokens/total": 13048800, "tokens/train_per_sec_per_gpu": 31.52, "tokens/trainable": 197109} +{"epoch": 1.6875, "grad_norm": 0.023997988551855087, "learning_rate": 1.600468845019576e-05, "loss": 0.0002723708748817444, "memory/device_reserved (GiB)": 36.36, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00027, "step": 432, "tokens/total": 13079392, "tokens/train_per_sec_per_gpu": 39.22, "tokens/trainable": 197607} +{"epoch": 1.69140625, "grad_norm": 0.03983060643076897, "learning_rate": 1.5860625757072092e-05, "loss": 0.0006240076618269086, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.95, "memory/max_allocated (GiB)": 33.95, "ppl": 1.00062, "step": 433, "tokens/total": 13109920, "tokens/train_per_sec_per_gpu": 33.34, "tokens/trainable": 198066} +{"epoch": 1.6953125, "grad_norm": 0.03856171295046806, "learning_rate": 1.571819181307116e-05, "loss": 0.0005622098105959594, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00056, "step": 434, "tokens/total": 13140064, "tokens/train_per_sec_per_gpu": 33.11, "tokens/trainable": 198484} +{"epoch": 1.69921875, "grad_norm": 0.07453715056180954, "learning_rate": 1.557739254545075e-05, "loss": 0.0008548495243303478, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00086, "step": 435, "tokens/total": 13170656, "tokens/train_per_sec_per_gpu": 35.03, "tokens/trainable": 198957} +{"epoch": 1.703125, "grad_norm": 0.0035155205987393856, "learning_rate": 1.543823381344311e-05, "loss": 5.2120005420874804e-05, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00005, "step": 436, "tokens/total": 13201056, "tokens/train_per_sec_per_gpu": 33.78, "tokens/trainable": 199405} +{"epoch": 1.70703125, "grad_norm": 0.9568848013877869, "learning_rate": 1.5300721408011114e-05, "loss": 0.00043475639540702105, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00043, "step": 437, "tokens/total": 13231584, "tokens/train_per_sec_per_gpu": 34.61, "tokens/trainable": 199888} +{"epoch": 1.7109375, "grad_norm": 0.010019529610872269, "learning_rate": 1.5164861051607254e-05, "loss": 0.000111720735731069, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00011, "step": 438, "tokens/total": 13261888, "tokens/train_per_sec_per_gpu": 35.46, "tokens/trainable": 200396} +{"epoch": 1.71484375, "grad_norm": 0.008281780406832695, "learning_rate": 1.5030658397935521e-05, "loss": 0.00017071102047339082, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00017, "step": 439, "tokens/total": 13291936, "tokens/train_per_sec_per_gpu": 33.34, "tokens/trainable": 200843} +{"epoch": 1.71875, "grad_norm": 0.06077976152300835, "learning_rate": 1.4898119031716104e-05, "loss": 0.0007504730019718409, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00075, "step": 440, "tokens/total": 13322224, "tokens/train_per_sec_per_gpu": 35.57, "tokens/trainable": 201315} +{"epoch": 1.72265625, "grad_norm": 0.12958590686321259, "learning_rate": 1.476724846845306e-05, "loss": 0.001083776238374412, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00108, "step": 441, "tokens/total": 13352640, "tokens/train_per_sec_per_gpu": 34.77, "tokens/trainable": 201782} +{"epoch": 1.7265625, "grad_norm": 0.04559873417019844, "learning_rate": 1.463805215420471e-05, "loss": 0.0005228023510426283, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.62, "memory/max_allocated (GiB)": 33.62, "ppl": 1.00052, "step": 442, "tokens/total": 13382560, "tokens/train_per_sec_per_gpu": 31.27, "tokens/trainable": 202222} +{"epoch": 1.73046875, "grad_norm": 0.02645108290016651, "learning_rate": 1.451053546535705e-05, "loss": 0.0003615481255110353, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00036, "step": 443, "tokens/total": 13412976, "tokens/train_per_sec_per_gpu": 36.81, "tokens/trainable": 202660} +{"epoch": 1.734375, "grad_norm": 0.021894006058573723, "learning_rate": 1.438470370840001e-05, "loss": 0.000229351528105326, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.64, "memory/max_allocated (GiB)": 33.64, "ppl": 1.00023, "step": 444, "tokens/total": 13442720, "tokens/train_per_sec_per_gpu": 31.51, "tokens/trainable": 203068} +{"epoch": 1.73828125, "grad_norm": 0.1398804783821106, "learning_rate": 1.4260562119706606e-05, "loss": 0.0013714064843952656, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00137, "step": 445, "tokens/total": 13472928, "tokens/train_per_sec_per_gpu": 34.97, "tokens/trainable": 203527} +{"epoch": 1.7421875, "grad_norm": 0.024557381868362427, "learning_rate": 1.413811586531508e-05, "loss": 0.00023480730305891484, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00023, "step": 446, "tokens/total": 13502992, "tokens/train_per_sec_per_gpu": 35.94, "tokens/trainable": 203994} +{"epoch": 1.74609375, "grad_norm": 0.021804476156830788, "learning_rate": 1.4017370040713884e-05, "loss": 0.00022578673087991774, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00023, "step": 447, "tokens/total": 13533184, "tokens/train_per_sec_per_gpu": 38.06, "tokens/trainable": 204430} +{"epoch": 1.75, "grad_norm": 0.23265990614891052, "learning_rate": 1.3898329670629645e-05, "loss": 0.002107386477291584, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00211, "step": 448, "tokens/total": 13563648, "tokens/train_per_sec_per_gpu": 36.07, "tokens/trainable": 204886} +{"epoch": 1.75390625, "grad_norm": 0.015779122710227966, "learning_rate": 1.3780999708818058e-05, "loss": 0.00014222922618500888, "memory/device_reserved (GiB)": 36.81, "memory/max_active (GiB)": 33.93, "memory/max_allocated (GiB)": 33.93, "ppl": 1.00014, "step": 449, "tokens/total": 13594032, "tokens/train_per_sec_per_gpu": 34.15, "tokens/trainable": 205356} +{"epoch": 1.7578125, "grad_norm": 0.09757973998785019, "learning_rate": 1.3665385037857758e-05, "loss": 0.0012260058429092169, "memory/device_reserved (GiB)": 35.69, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00123, "step": 450, "tokens/total": 13624368, "tokens/train_per_sec_per_gpu": 31.13, "tokens/trainable": 205782} +{"epoch": 1.76171875, "grad_norm": 0.05971763655543327, "learning_rate": 1.3551490468947126e-05, "loss": 0.0006241232040338218, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00062, "step": 451, "tokens/total": 13654768, "tokens/train_per_sec_per_gpu": 32.1, "tokens/trainable": 206219} +{"epoch": 1.765625, "grad_norm": 0.009018263779580593, "learning_rate": 1.3439320741704075e-05, "loss": 5.874139242223464e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00006, "step": 452, "tokens/total": 13684960, "tokens/train_per_sec_per_gpu": 36.29, "tokens/trainable": 206677} +{"epoch": 1.76953125, "grad_norm": 0.011312981136143208, "learning_rate": 1.3328880523968808e-05, "loss": 8.52083321660757e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00009, "step": 453, "tokens/total": 13715488, "tokens/train_per_sec_per_gpu": 37.8, "tokens/trainable": 207160} +{"epoch": 1.7734375, "grad_norm": 0.0022447824012488127, "learning_rate": 1.3220174411609587e-05, "loss": 3.376982203917578e-05, "memory/device_reserved (GiB)": 35.7, "memory/max_active (GiB)": 33.86, "memory/max_allocated (GiB)": 33.86, "ppl": 1.00003, "step": 454, "tokens/total": 13745856, "tokens/train_per_sec_per_gpu": 36.25, "tokens/trainable": 207605} +{"epoch": 1.77734375, "grad_norm": 0.017390765249729156, "learning_rate": 1.3113206928331471e-05, "loss": 0.00019571901066228747, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.0002, "step": 455, "tokens/total": 13776192, "tokens/train_per_sec_per_gpu": 33.43, "tokens/trainable": 208047} +{"epoch": 1.78125, "grad_norm": 0.048852503299713135, "learning_rate": 1.300798252548806e-05, "loss": 0.0005775241879746318, "memory/device_reserved (GiB)": 35.78, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00058, "step": 456, "tokens/total": 13806496, "tokens/train_per_sec_per_gpu": 35.44, "tokens/trainable": 208476} +{"epoch": 1.78515625, "grad_norm": 0.3256935179233551, "learning_rate": 1.2904505581896265e-05, "loss": 0.0016588940052315593, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00166, "step": 457, "tokens/total": 13836768, "tokens/train_per_sec_per_gpu": 36.03, "tokens/trainable": 208970} +{"epoch": 1.7890625, "grad_norm": 0.06032567098736763, "learning_rate": 1.2802780403654082e-05, "loss": 0.000697342911735177, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.0007, "step": 458, "tokens/total": 13867152, "tokens/train_per_sec_per_gpu": 35.31, "tokens/trainable": 209456} +{"epoch": 1.79296875, "grad_norm": 0.007340370211750269, "learning_rate": 1.2702811223961408e-05, "loss": 3.1743944418849424e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00003, "step": 459, "tokens/total": 13897360, "tokens/train_per_sec_per_gpu": 34.15, "tokens/trainable": 209907} +{"epoch": 1.796875, "grad_norm": 0.16073353588581085, "learning_rate": 1.2604602202943861e-05, "loss": 0.001599560840986669, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.91, "memory/max_allocated (GiB)": 33.91, "ppl": 1.0016, "step": 460, "tokens/total": 13927792, "tokens/train_per_sec_per_gpu": 34.16, "tokens/trainable": 210366} +{"epoch": 1.80078125, "grad_norm": 0.5971834659576416, "learning_rate": 1.2508157427479686e-05, "loss": 0.002240244299173355, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.65, "memory/max_allocated (GiB)": 33.65, "ppl": 1.00224, "step": 461, "tokens/total": 13957856, "tokens/train_per_sec_per_gpu": 32.1, "tokens/trainable": 210835} +{"epoch": 1.8046875, "grad_norm": 0.15249626338481903, "learning_rate": 1.2413480911029655e-05, "loss": 0.0012240726500749588, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.38, "memory/max_allocated (GiB)": 33.38, "ppl": 1.00122, "step": 462, "tokens/total": 13986080, "tokens/train_per_sec_per_gpu": 30.91, "tokens/trainable": 211259} +{"epoch": 1.80859375, "grad_norm": 0.005158649291843176, "learning_rate": 1.2320576593470082e-05, "loss": 3.6364894185680896e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.79, "memory/max_allocated (GiB)": 33.79, "ppl": 1.00004, "step": 463, "tokens/total": 14016288, "tokens/train_per_sec_per_gpu": 35.65, "tokens/trainable": 211721} +{"epoch": 1.8125, "grad_norm": 0.11264989525079727, "learning_rate": 1.2229448340928828e-05, "loss": 0.001997420797124505, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.74, "memory/max_allocated (GiB)": 33.74, "ppl": 1.002, "step": 464, "tokens/total": 14046288, "tokens/train_per_sec_per_gpu": 29.59, "tokens/trainable": 212159} +{"epoch": 1.81640625, "grad_norm": 0.020843395963311195, "learning_rate": 1.2140099945624458e-05, "loss": 0.0001486139663029462, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00015, "step": 465, "tokens/total": 14076512, "tokens/train_per_sec_per_gpu": 33.17, "tokens/trainable": 212644} +{"epoch": 1.8203125, "grad_norm": 0.0014335883315652609, "learning_rate": 1.205253512570841e-05, "loss": 2.982833029818721e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00003, "step": 466, "tokens/total": 14106608, "tokens/train_per_sec_per_gpu": 33.38, "tokens/trainable": 213084} +{"epoch": 1.82421875, "grad_norm": 0.17584311962127686, "learning_rate": 1.1966757525110255e-05, "loss": 0.0020335109438747168, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00204, "step": 467, "tokens/total": 14136864, "tokens/train_per_sec_per_gpu": 32.05, "tokens/trainable": 213523} +{"epoch": 1.828125, "grad_norm": 0.012916951440274715, "learning_rate": 1.1882770713386095e-05, "loss": 0.00013626206782646477, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.73, "memory/max_allocated (GiB)": 33.73, "ppl": 1.00014, "step": 468, "tokens/total": 14166896, "tokens/train_per_sec_per_gpu": 35.03, "tokens/trainable": 213963} +{"epoch": 1.83203125, "grad_norm": 0.13297978043556213, "learning_rate": 1.180057818556998e-05, "loss": 0.0010875939624384046, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00109, "step": 469, "tokens/total": 14197568, "tokens/train_per_sec_per_gpu": 37.4, "tokens/trainable": 214431} +{"epoch": 1.8359375, "grad_norm": 0.0005420309607870877, "learning_rate": 1.1720183362028494e-05, "loss": 1.2743539627990685e-05, "memory/device_reserved (GiB)": 35.82, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00001, "step": 470, "tokens/total": 14225984, "tokens/train_per_sec_per_gpu": 36.07, "tokens/trainable": 214865} +{"epoch": 1.83984375, "grad_norm": 0.3001365065574646, "learning_rate": 1.1641589588318387e-05, "loss": 0.0015361867845058441, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00154, "step": 471, "tokens/total": 14256512, "tokens/train_per_sec_per_gpu": 35.98, "tokens/trainable": 215358} +{"epoch": 1.84375, "grad_norm": 0.014027237892150879, "learning_rate": 1.1564800135047418e-05, "loss": 0.00015228531265165657, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00015, "step": 472, "tokens/total": 14286640, "tokens/train_per_sec_per_gpu": 33.38, "tokens/trainable": 215800} +{"epoch": 1.84765625, "grad_norm": 0.08512959629297256, "learning_rate": 1.148981819773816e-05, "loss": 0.000548110285308212, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00055, "step": 473, "tokens/total": 14317024, "tokens/train_per_sec_per_gpu": 35.29, "tokens/trainable": 216245} +{"epoch": 1.8515625, "grad_norm": 0.004200903698801994, "learning_rate": 1.1416646896695086e-05, "loss": 4.4981639803154394e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00004, "step": 474, "tokens/total": 14347088, "tokens/train_per_sec_per_gpu": 33.83, "tokens/trainable": 216685} +{"epoch": 1.85546875, "grad_norm": 0.4064914584159851, "learning_rate": 1.1345289276874717e-05, "loss": 0.002648166147992015, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00265, "step": 475, "tokens/total": 14377152, "tokens/train_per_sec_per_gpu": 31.31, "tokens/trainable": 217092} +{"epoch": 1.859375, "grad_norm": 0.0013920688070356846, "learning_rate": 1.1275748307758873e-05, "loss": 1.866836828412488e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00002, "step": 476, "tokens/total": 14407568, "tokens/train_per_sec_per_gpu": 38.27, "tokens/trainable": 217582} +{"epoch": 1.86328125, "grad_norm": 0.0018797408556565642, "learning_rate": 1.1208026883231147e-05, "loss": 2.958982076961547e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00003, "step": 477, "tokens/total": 14437792, "tokens/train_per_sec_per_gpu": 36.2, "tokens/trainable": 218049} +{"epoch": 1.8671875, "grad_norm": 0.10846851021051407, "learning_rate": 1.1142127821456433e-05, "loss": 0.000766873883549124, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00077, "step": 478, "tokens/total": 14468400, "tokens/train_per_sec_per_gpu": 33.06, "tokens/trainable": 218522} +{"epoch": 1.87109375, "grad_norm": 0.08651807904243469, "learning_rate": 1.1078053864763674e-05, "loss": 0.0007375412969850004, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.83, "memory/max_allocated (GiB)": 33.83, "ppl": 1.00074, "step": 479, "tokens/total": 14498416, "tokens/train_per_sec_per_gpu": 33.32, "tokens/trainable": 218970} +{"epoch": 1.875, "grad_norm": 0.19086171686649323, "learning_rate": 1.1015807679531756e-05, "loss": 0.002163141267374158, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.98, "memory/max_allocated (GiB)": 33.98, "ppl": 1.00217, "step": 480, "tokens/total": 14528976, "tokens/train_per_sec_per_gpu": 34.62, "tokens/trainable": 219411} +{"epoch": 1.87890625, "grad_norm": 0.0037188229616731405, "learning_rate": 1.0955391856078528e-05, "loss": 3.280482633272186e-05, "memory/device_reserved (GiB)": 35.92, "memory/max_active (GiB)": 33.78, "memory/max_allocated (GiB)": 33.78, "ppl": 1.00003, "step": 481, "tokens/total": 14558992, "tokens/train_per_sec_per_gpu": 29.98, "tokens/trainable": 219837} +{"epoch": 1.8828125, "grad_norm": 0.0028978160116821527, "learning_rate": 1.0896808908553007e-05, "loss": 2.934904841822572e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00003, "step": 482, "tokens/total": 14589488, "tokens/train_per_sec_per_gpu": 35.61, "tokens/trainable": 220345} +{"epoch": 1.88671875, "grad_norm": 0.030241984874010086, "learning_rate": 1.0840061274830763e-05, "loss": 0.0002913741336669773, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.8, "memory/max_allocated (GiB)": 33.8, "ppl": 1.00029, "step": 483, "tokens/total": 14619584, "tokens/train_per_sec_per_gpu": 33.58, "tokens/trainable": 220775} +{"epoch": 1.890625, "grad_norm": 0.0025161579251289368, "learning_rate": 1.0785151316412473e-05, "loss": 2.7965787012362853e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00003, "step": 484, "tokens/total": 14649760, "tokens/train_per_sec_per_gpu": 35.58, "tokens/trainable": 221217} +{"epoch": 1.89453125, "grad_norm": 0.014441153965890408, "learning_rate": 1.0732081318325639e-05, "loss": 8.626213093521073e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00009, "step": 485, "tokens/total": 14679920, "tokens/train_per_sec_per_gpu": 32.07, "tokens/trainable": 221666} +{"epoch": 1.8984375, "grad_norm": 0.016399968415498734, "learning_rate": 1.0680853489029501e-05, "loss": 0.00012868674821220338, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00013, "step": 486, "tokens/total": 14710432, "tokens/train_per_sec_per_gpu": 31.95, "tokens/trainable": 222110} +{"epoch": 1.90234375, "grad_norm": 0.021996503695845604, "learning_rate": 1.0631469960323152e-05, "loss": 0.00015319210069719702, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.67, "memory/max_allocated (GiB)": 33.67, "ppl": 1.00015, "step": 487, "tokens/total": 14740416, "tokens/train_per_sec_per_gpu": 32.47, "tokens/trainable": 222542} +{"epoch": 1.90625, "grad_norm": 0.002610082970932126, "learning_rate": 1.0583932787256783e-05, "loss": 1.0041851055575535e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00001, "step": 488, "tokens/total": 14770832, "tokens/train_per_sec_per_gpu": 36.81, "tokens/trainable": 223035} +{"epoch": 1.91015625, "grad_norm": 0.0016527038533240557, "learning_rate": 1.0538243948046206e-05, "loss": 1.9269999029347673e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.77, "memory/max_allocated (GiB)": 33.77, "ppl": 1.00002, "step": 489, "tokens/total": 14801104, "tokens/train_per_sec_per_gpu": 32.27, "tokens/trainable": 223488} +{"epoch": 1.9140625, "grad_norm": 0.0007155581843107939, "learning_rate": 1.0494405343990523e-05, "loss": 6.305221177171916e-06, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.76, "memory/max_allocated (GiB)": 33.76, "ppl": 1.00001, "step": 490, "tokens/total": 14831216, "tokens/train_per_sec_per_gpu": 29.86, "tokens/trainable": 223932} +{"epoch": 1.91796875, "grad_norm": 0.013574068434536457, "learning_rate": 1.0452418799392985e-05, "loss": 0.00013808548101224005, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00014, "step": 491, "tokens/total": 14861488, "tokens/train_per_sec_per_gpu": 33.39, "tokens/trainable": 224389} +{"epoch": 1.921875, "grad_norm": 0.21587051451206207, "learning_rate": 1.0412286061485102e-05, "loss": 0.001721705892123282, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.01, "memory/max_allocated (GiB)": 34.01, "ppl": 1.00172, "step": 492, "tokens/total": 14891824, "tokens/train_per_sec_per_gpu": 35.65, "tokens/trainable": 224871} +{"epoch": 1.92578125, "grad_norm": 0.4014979600906372, "learning_rate": 1.03740088003539e-05, "loss": 0.016092870384454727, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.01622, "step": 493, "tokens/total": 14922192, "tokens/train_per_sec_per_gpu": 35.16, "tokens/trainable": 225352} +{"epoch": 1.9296875, "grad_norm": 0.0035919942893087864, "learning_rate": 1.0337588608872463e-05, "loss": 5.178125138627365e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.84, "memory/max_allocated (GiB)": 33.84, "ppl": 1.00005, "step": 494, "tokens/total": 14952624, "tokens/train_per_sec_per_gpu": 33.3, "tokens/trainable": 225801} +{"epoch": 1.93359375, "grad_norm": 0.0015177514869719744, "learning_rate": 1.0303027002633622e-05, "loss": 1.5969462765497155e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.81, "memory/max_allocated (GiB)": 33.81, "ppl": 1.00002, "step": 495, "tokens/total": 14982704, "tokens/train_per_sec_per_gpu": 33.05, "tokens/trainable": 226240} +{"epoch": 1.9375, "grad_norm": 0.0016166599234566092, "learning_rate": 1.0270325419886884e-05, "loss": 2.797747583827004e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.9, "memory/max_allocated (GiB)": 33.9, "ppl": 1.00003, "step": 496, "tokens/total": 15013024, "tokens/train_per_sec_per_gpu": 35.32, "tokens/trainable": 226712} +{"epoch": 1.94140625, "grad_norm": 0.007683792617172003, "learning_rate": 1.0239485221478599e-05, "loss": 9.649172716308385e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.0001, "step": 497, "tokens/total": 15043520, "tokens/train_per_sec_per_gpu": 32.28, "tokens/trainable": 227135} +{"epoch": 1.9453125, "grad_norm": 0.0021688074339181185, "learning_rate": 1.0210507690795292e-05, "loss": 3.7211153539828956e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.87, "memory/max_allocated (GiB)": 33.87, "ppl": 1.00004, "step": 498, "tokens/total": 15073824, "tokens/train_per_sec_per_gpu": 38.25, "tokens/trainable": 227600} +{"epoch": 1.94921875, "grad_norm": 0.028605012223124504, "learning_rate": 1.0183394033710305e-05, "loss": 0.0003017223789356649, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.0003, "step": 499, "tokens/total": 15104128, "tokens/train_per_sec_per_gpu": 37.21, "tokens/trainable": 228076} +{"epoch": 1.953125, "grad_norm": 0.020242059603333473, "learning_rate": 1.0158145378533583e-05, "loss": 6.54559480608441e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.82, "memory/max_allocated (GiB)": 33.82, "ppl": 1.00007, "step": 500, "tokens/total": 15134480, "tokens/train_per_sec_per_gpu": 31.41, "tokens/trainable": 228526} +{"epoch": 1.95703125, "grad_norm": 0.0019631576724350452, "learning_rate": 1.0134762775964726e-05, "loss": 2.9010820071562193e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.92, "memory/max_allocated (GiB)": 33.92, "ppl": 1.00003, "step": 501, "tokens/total": 15164832, "tokens/train_per_sec_per_gpu": 35.76, "tokens/trainable": 229005} +{"epoch": 1.9609375, "grad_norm": 0.05687836930155754, "learning_rate": 1.0113247199049278e-05, "loss": 0.0003191005380358547, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.03, "memory/max_allocated (GiB)": 34.03, "ppl": 1.00032, "step": 502, "tokens/total": 15195616, "tokens/train_per_sec_per_gpu": 31.5, "tokens/trainable": 229438} +{"epoch": 1.96484375, "grad_norm": 0.04075845330953598, "learning_rate": 1.0093599543138205e-05, "loss": 0.00027428093017078936, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00027, "step": 503, "tokens/total": 15226096, "tokens/train_per_sec_per_gpu": 33.33, "tokens/trainable": 229885} +{"epoch": 1.96875, "grad_norm": 0.01305411383509636, "learning_rate": 1.0075820625850675e-05, "loss": 0.00012350856559351087, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.96, "memory/max_allocated (GiB)": 33.96, "ppl": 1.00012, "step": 504, "tokens/total": 15256512, "tokens/train_per_sec_per_gpu": 39.53, "tokens/trainable": 230389} +{"epoch": 1.97265625, "grad_norm": 0.08244533091783524, "learning_rate": 1.0059911187040013e-05, "loss": 0.0005182232707738876, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.97, "memory/max_allocated (GiB)": 33.97, "ppl": 1.00052, "step": 505, "tokens/total": 15287072, "tokens/train_per_sec_per_gpu": 38.08, "tokens/trainable": 230885} +{"epoch": 1.9765625, "grad_norm": 0.001344834454357624, "learning_rate": 1.0045871888762893e-05, "loss": 1.9871291442541406e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.85, "memory/max_allocated (GiB)": 33.85, "ppl": 1.00002, "step": 506, "tokens/total": 15317296, "tokens/train_per_sec_per_gpu": 34.79, "tokens/trainable": 231335} +{"epoch": 1.98046875, "grad_norm": 0.0459468699991703, "learning_rate": 1.003370331525184e-05, "loss": 0.00030574080301448703, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00031, "step": 507, "tokens/total": 15347904, "tokens/train_per_sec_per_gpu": 36.96, "tokens/trainable": 231824} +{"epoch": 1.984375, "grad_norm": 0.0023596102837473154, "learning_rate": 1.002340597289085e-05, "loss": 2.902157575590536e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.33, "memory/max_allocated (GiB)": 33.33, "ppl": 1.00003, "step": 508, "tokens/total": 15376144, "tokens/train_per_sec_per_gpu": 32.08, "tokens/trainable": 232241} +{"epoch": 1.98828125, "grad_norm": 0.09540333598852158, "learning_rate": 1.0014980290194387e-05, "loss": 0.0011963924625888467, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 34.02, "memory/max_allocated (GiB)": 34.02, "ppl": 1.0012, "step": 509, "tokens/total": 15406976, "tokens/train_per_sec_per_gpu": 37.83, "tokens/trainable": 232742} +{"epoch": 1.9921875, "grad_norm": 0.0034284063149243593, "learning_rate": 1.0008426617789489e-05, "loss": 4.3011394154746085e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.94, "memory/max_allocated (GiB)": 33.94, "ppl": 1.00004, "step": 510, "tokens/total": 15437344, "tokens/train_per_sec_per_gpu": 36.66, "tokens/trainable": 233231} +{"epoch": 1.99609375, "grad_norm": 0.01952311210334301, "learning_rate": 1.0003745228401215e-05, "loss": 0.0001366769429296255, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.88, "memory/max_allocated (GiB)": 33.88, "ppl": 1.00014, "step": 511, "tokens/total": 15467712, "tokens/train_per_sec_per_gpu": 30.86, "tokens/trainable": 233655} +{"epoch": 2.0, "grad_norm": 0.004961141850799322, "learning_rate": 1.0000936316841296e-05, "loss": 8.935239748097956e-05, "memory/device_reserved (GiB)": 36.15, "memory/max_active (GiB)": 33.89, "memory/max_allocated (GiB)": 33.89, "ppl": 1.00009, "step": 512, "tokens/total": 15498240, "tokens/train_per_sec_per_gpu": 36.78, "tokens/trainable": 234124}