diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..3fff3d3496de6a76c8c1ea6f6a7b4dd4ccece543 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eecc5243a2ed93ac65bea8de8373ff27e5208b62b6b17d31b14c9a68eb37394b +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..13dce95a8d1c26b030f7ecaf731d168b9d13e5c1 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:29b34641c29769169a1ed6b4afd58e9e307067cada1e9e88c72966e0144f438d +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..98dbeeb3f080b71feaff938c500faf89b48b90d8 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_004500_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c13ffbd7e162bf3c9a022393fd3901c55e8db1cf896cb08c0bdee97c2b6dc83c +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..7b861d769274333974f8e33297c8f1eb37e321e0 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3ff0cc947e18af1a04987dc2f64fa0a29d22418d34efdff24027a1798c7cea94 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..9b0368f4cae03c7c2ca414b502cd89ab8eb5cd37 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:085b891a2d75abe7ff9687ec43f89c444604eb83bbf0d5f903b21675eb51160f +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..d042f60072e4692b0a0637901293c48e6d3a5166 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7f20218827152f212517747857afab2b7349aa5a043d96c23f296a5907d717ba +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..307c8b39a345d1aaf4693a2aa3bfa6553df18ed4 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005000_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b94746b3fa690d69a9ba3e5cbde7c7da3e1f0aabb50b73c9f3b1326f4fc38743 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..97b5eb307d67a3ddc9f52bf474524076744f2983 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:225af4d6d4d822abbce34e63d621eac376be5b8acf857ba8997e36582762573a +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..08c348506761b7a8f50b88e53f0cbeb838aa5249 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bda69aaa71c4ac52b0c4658c58b7d07f8671608af2f71a512e983da688149ad6 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..5a4e2e942eef55872dbc96281ed4ff32b842b2ad --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c39f5feb695f84f82218defbe6c535b506173b550fd3ae6fe0caa20278817f3f +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..c61757cd5c711e9694cb916d45a9e3c8199cf222 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_005500_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b806a61d71740cbb0e2b088ea4c42165334ecd9863494c27a343b548b0d84326 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..fd2f5d56db044859c812df5fd2bbfd059d3bd21d --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f7e168383bc13bbc56df91c21f3d46e835001ed592f90f849af53993f9c965a3 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..9d7907599f9c8c5ceaf91e5db771254a97b510f9 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a068ed26f38b48c5c7ef70a1771da690ab6cdd1749d03f902277b5b2a553f94e +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..a7b95224bd425230871a29870c5278c70400b007 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f63a2df3dbbe97e7bbf952c0fc0eb732423d49fe7e994c9b632e1342d9ac8868 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..972682cada1c58e72a6fcdf9f0212fbf56440dc9 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006000_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:60613cd93ddac30dd28676e47ae16588c2ad85aa09b9bd4f63f17a5ad73dfc07 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..9f93afed1bd0ecb6105660308bc2a5a3d5ccbd01 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7a5fd4763414c513d80406c35ecdf9f26ec5d1d9d27a676e39dfd01c8e6992b7 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..4bef645d0b66dfee1f1d7fd0d2b2f4e9582d351b --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b21033cf7c8929fe01be78ca45b157751d65c7dff5cb57940226319a9d76d9c +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..79f3c8fe08e743ead381bcfad755104938ba57a1 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:99ac47912efb2feda2492d9f772370071b090843ee13e55a08e1e31d7068926a +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..9d412edbc4b071b6222f9bb84677d8c05361eb31 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_006500_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:10a5f118f4f3537eaf2230fbeef6fe6e9298958787e807116fadf09f5fc0e04e +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..8ae39270d361a274d0537b6f5cea28b86b715ba7 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d27d602d30b1a4d7aa9605fa175739fb161b32e02909ec30c231e3fde3be8264 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..be238b1077c4b9f3007287b37c203d663d72a251 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e638779e6afd341c5453a5de23e7c834f305837bd70cc5a0fc12b277cd7ee9fc +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..7dd7b85740cb95e60a4008b72b3aa3077437a4ce --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a1afe22caa4f2819cfc397846b74b8c26272f6266a16746d31f3a87bac3a5c37 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..06a116d9c933a6dc0e2dbe3234b5816cd49a810e --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007000_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:acbcb96c068e6d855d7619f22c370140f31a894c4d23f105c7522214d8a51ae4 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b1537d401d561c880d1faa625adc81be8aafad24 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7c8beda301552c587b21e718b3f0eb383006dbb232a26d5951b9132153346c4d +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..7790f7ac35002bbe773082b795089f511cc02e95 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7db4c7619d2c479f1b6a41805b4fc74865c7f4563cf71d8105af4ecc6f9ffd27 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..42a78a54b5c4468dc66fbc6b0b8b3877c3bdfaee --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cdc736fbeef52ca1bfe7e975b93ad550bfc35129ba2e25eddad9ecf59d3fc3bf +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..d05e178495cd94a09a04506f0b343e69b808a9ba --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_007500_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:30523b0f3a84b56579b1843151587d357fbae481de2e0885f92b484ed6356e2e +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..2858604c969f00c2fddb45e4209e5c0802aa9585 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:11ff33d778dc9ba30dc3fda9174bfe1b73fd251370a901f1e396ab7de520028e +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..20b10659b1b4b4e09e6187fb9cac0b7e7b1d852c --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fa40a10a4058f06d01d3393321e236973fa78f2d137d1e02d7ca4c119b7305e6 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..d842272f70f6f28b6c98f6a116a587c40673b354 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:22d887c0fc65e71df1f3d74b5ea3656ca860658d8249aa8cac96421e89a7a835 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..79f86732149bf73e5dfb1bb3886e49c01c8ea6e8 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008000_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fc131b1f33fa81372bcc1ecb759f2b7ddf2acf55652b07e1a14b26b96a909e73 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..5670db0bbfcac33c0e5aa71e025e54e9d33a34e7 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ef67bbc1c6e6f73cf3ccfe50d3df3e916b2cf913a2ef1fb6a14123e784e4bc13 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank1.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank1.pt new file mode 100644 index 0000000000000000000000000000000000000000..230997d0dc5116c8c5be6dda4c3b08009e4236d7 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank1.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:839d78c2d9e329f91167b3ca2a706ba931d4fc946101826a6bbac664b3c366b1 +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank2.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank2.pt new file mode 100644 index 0000000000000000000000000000000000000000..5a6ae573964c5e34aaad1fdc5b22e1df2004bd78 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank2.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7987c59cd7ef08458fcdf58f4e737501e3adf2e5a32cd0e44015ba70ee68d29b +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank3.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank3.pt new file mode 100644 index 0000000000000000000000000000000000000000..7ec76ac70790641f6de6c4aa61c6124217a02a89 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints/optim_008352_rank3.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b5c08a5e6fd49981ac6887b43d4d835742ef20e8eefa060670635dedc0c69de +size 1434913073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json new file mode 100644 index 0000000000000000000000000000000000000000..83f912760475957e23448e8669d54495f0cc358d --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/config.json @@ -0,0 +1,63 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean-1930s", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 180, + "download_workers": 4 + }, + "tokenizer": { + "mode": "reuse", + "source_experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 24, + "scaling_params": 729810624, + "target_param_data_ratio": 12, + "max_seq_len": 4096, + "window_pattern": "SSSL", + "device_batch_size": 8, + "total_batch_size": 1048576, + "fp8": true, + "fp8_recipe": "tensorwise", + "seed": 42, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano", + "keep_local_checkpoints": 1 + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "group": "clean1930s-d24", + "tags": [ + "think-dataset-clean-1930s", + "d24", + "ratio12", + "ctx4096", + "sssl", + "fp8" + ] + }, + "config_fingerprint": "166e7e695c1b536a", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/core.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..9dd677469d98b2c43c74fc93887dafa421454807 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/core.json @@ -0,0 +1,56 @@ +{ + "model": "base_model (step 8352)", + "step": 8352, + "bpb": {}, + "core_metric": 0.13542864491271994, + "core_results": { + "hellaswag_zeroshot": 0.32364070415496826, + "jeopardy": 0.0037789323832839727, + "bigbench_qa_wikidata": 0.2516116201877594, + "arc_easy": 0.41750839352607727, + "arc_challenge": 0.239761084318161, + "copa": 0.6100000143051147, + "commonsense_qa": 0.3071253001689911, + "piqa": 0.5805222988128662, + "openbook_qa": 0.2580000162124634, + "lambada_openai": 0.35338637232780457, + "hellaswag": 0.32911768555641174, + "winograd": 0.6703296899795532, + "winogrande": 0.49408048391342163, + "bigbench_dyck_languages": 0.12700000405311584, + "agi_eval_lsat_ar": 0.260869562625885, + "bigbench_cs_algorithms": 0.38333332538604736, + "bigbench_operators": 0.11428572237491608, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.114096499979496, + "coqa": 0.16610296070575714, + "boolq": 0.5969418883323669, + "bigbench_language_identification": 0.25699999928474426 + }, + "centered_results": { + "hellaswag_zeroshot": 0.09818760553995769, + "jeopardy": 0.0037789323832839727, + "bigbench_qa_wikidata": 0.2516116201877594, + "arc_easy": 0.22334452470143637, + "arc_challenge": -0.013651887575785318, + "copa": 0.2200000286102295, + "commonsense_qa": 0.13390662521123883, + "piqa": 0.16104459762573242, + "openbook_qa": 0.010666688283284506, + "lambada_openai": 0.35338637232780457, + "hellaswag": 0.105490247408549, + "winograd": 0.34065937995910645, + "winogrande": -0.011839032173156738, + "bigbench_dyck_languages": 0.12700000405311584, + "agi_eval_lsat_ar": 0.07608695328235625, + "bigbench_cs_algorithms": 0.38333332538604736, + "bigbench_operators": 0.11428572237491608, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.114096499979496, + "coqa": 0.16610296070575714, + "boolq": -0.060679241230613294, + "bigbench_language_identification": 0.1826182610393226 + }, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/samples.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..2b91acc601be03a4af30318151f8cf79249cf54d --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 8352)", + "step": 8352, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is the city of Paris, which is the capital of France. The capital of France" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is Au, and the symbol of silver is Ag. The symbol of gold is Au" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Friday, then tomorrow will be Saturday. If yesterday was" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is cold.\n\nThe opposite of cold is hot.\n\nThe opposite of hot is cold.\n\n" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Earth, Mars, Jupiter, Saturn, Uranus," + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is a dark brown, with a tinge of red. It is a very pretty color" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is 13, and the equation is 13x + 3 = 13" + } + ], + "unconditioned_samples": [ + "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Psychological Bulletin,\" of the University of Chicago, during the past summer; the temporary arrangements for its printing prevented its meeting the demand for it. Fair play in the matter of results in the revision of Shakespeare's plays has not been given in this field in this company, Dunning, Uzziel, and others engaging to contribute whatever they have to say on the revision and construction of even a very few plays, and the magazine enterprise to furnish more plays of a classic nature, although comprising many dramas of merit and of the utmost possible variety, is by no means of sufficient scope and interest", + "<|bos|>\n\nARTON LIBRARY OF THE DEZA SUPERINTENDENTS AND BOOKKEEPERS\n\nThe following Outline of Business\n\nthe State or Territory in which\n\nFoss work is to be carried on.\n\nPLAN OF BUSINESS Department 1. Traveling\n\n2. Records, coupons, checks, mercantile accounts, vouchers, &c.\n\n3. General office\n\n4. Subsidiary agencies. Persons dealing with department men.\n\nPlanwork.\n\nPlan.\n\n1. Plans for the method of review of machinery, etc. Uncertainty.\n\n2. Plans for securing uniformity of practice over a large field", + "<|bos|>!\n\nIt's the coss is mighty wonderous weak Thefe minutes yhere hain marry teef The tryfector\n\nOf my fandd in Life who ought to hast The pist o' a President\n\nApril 10. J,\"\n\nAmong the Arms in cison\n\nThe Arms of the President *The Arms of Washington\n\nOn October 5, (?) in another and more noticeable storm-swept plagiarism, the President bore the baptismal emblem of the United States; on the way to Washington, January 1, 1799, occurred the cancel that has made possible the interpolations of the following", + "<|bos|> ministered unto Him with great joy.\"\n\nXX. 1. CHIRIT OF THE BLESSED.\n\nThe full contrast to this joy is the hunger that followed when He left His throne in heaven, ascended, and took the place of the \"Name\" lifted up on Calvary.\n\n2. The Fulfilment of the Tabernacle: In our imagination let 50 Enter: first, the Church universal: prayer, glorification, rejoicing, nothing but an eternal army and the eternal joy which forms one side of this-and well! for consequently this is that infant Church which a voice descended, and the Redeemer Sunday broke through the wall that", + "<|bos|>eed\n\nNew York paid\n\nEach separate $1.25 and i piece by akely ried in Watts averaged 75 cents.\n\nRetail price 50 cents-No longer find it worth while to order second-hand books. Already considerable order great publishing interests are sending agents to England or Scotland to our mmission depot expressly to collect the $1.25 per set, on which income is made over to us. The influens can gain fifteen per cent, and this can be taken care of at once as it is here almost never debited. Its great advantages that has the most of A. L. Burlingame's", + "<|bos|>PREFACE\n\nWHEN PERSHAD was about to issue and was being attacked by the fanatical cuttlefish of Northern India, some weeks before at least the Calcutta of 1862 stood face to face with a panic-crazed gale, the doctors began one day on their platform to advise the world how they were going to appease the spirit of the Fish-god. The Elders and the Lady-Seconds criticized emphatically the pros and cons, and the whole city stared.\n\nOne man said: \"I am Oriental; the people of this English country do not believe in all this agnostic nonsense, and the skies when they are", + "<|bos|>iman have published,\n\nTo be completed in about 100 Monthly Parts, each illustrated by a Steel Engraving, a Work consisting of a Library of Original 180 Engravings, 3 vols. 8vo. \u00a35. 5s. strongly bound in cloth, as a Work for convenient\n\nSale by STEEL Engraved by W. L. STEENERSON, each neatly illustrated by a Steel Engraving; the Plates are in a very high state of\n\nPerfection, and warranted to give a correct Image of each Novel.\n\nThe Work is published in Monthly Parts to contain Historical and Biographical Sketches of the principal", + "<|bos|>MYSTER LIBRARY.\n\n3 2044 009 603 166 fidh\n\nQUE\n\n7808, W" + ] +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/val_bpb.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..f488ffe394f042d6c23f4e7edd1b3cf593fc9321 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 8352)", + "step": 8352, + "bpb": { + "val": 0.8435203902195944 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/run.json new file mode 100644 index 0000000000000000000000000000000000000000..255cd14f4914c9569b6ce62fb3ed77dfeb8b012f --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "stage": "base", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "166e7e695c1b536a", + "wandb_run_id": "e69c1e59", + "created_at": 1784557543 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/config.json new file mode 100644 index 0000000000000000000000000000000000000000..fc2b5b91b9afdcdf7b0dfcfb9e7621733d063c85 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/config.json @@ -0,0 +1,36 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "nanochat-default-v1", + "data": { + "recipe": "nanochat-default", + "mmlu_epochs": 3, + "gsm8k_epochs": 4 + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "eval_every": -1, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": 200 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": false, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "clean1930s-d24-r12", + "tags": [ + "sft", + "smoltalk", + "mmlu3", + "gsm8k4" + ] + }, + "config_fingerprint": "103d7c3526aa35c0", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/run.json new file mode 100644 index 0000000000000000000000000000000000000000..c04b45d5fdfbfc4203f0f441cf98173107e575e0 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-nanochat-default-v1", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352, + "branch_parent_step": null, + "config_fingerprint": "103d7c3526aa35c0", + "wandb_run_id": null, + "created_at": 1786512279 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/meta_000007.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/meta_000007.json new file mode 100644 index 0000000000000000000000000000000000000000..c376b67eb26d26eca43d12d72a50e81853709ea1 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/meta_000007.json @@ -0,0 +1,103 @@ +{ + "step": 7, + "training_complete": true, + "val_bpb": 0.7205382880492125, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic", + "wandb_run_id": "0837926f", + "wandb_group": "think-d12", + "wandb_tags": "sft,pre1930,ratio20", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic", + "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "60aa7005265901858554bb7ca98cd788de721e4a", + "load_optimizer": 1, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.0, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": -1, + "eval_tokens": 20971520, + "chatcore_every": -1, + "chatcore_max_cat": -1, + "chatcore_max_sample": 24, + "save_every": -1, + "recipe": "pre1930", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-authentic", + "data": { + "recipe": "pre1930", + "pre1930_epochs": 5 + }, + "training": { + "num_iterations": -1, + "device_batch_size": 8, + "eval_every": -1, + "chatcore_every": -1, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "pre1930", + "ratio20" + ] + }, + "config_fingerprint": "d3378357cef17359", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "d3378357cef17359" + }, + "loop_state": { + "step": 7, + "total_training_time": 0.0, + "min_val_bpb": 0.7205382880492125, + "smooth_train_loss": 1.187201474390888, + "mfu": 35.33209107765201, + "tok_per_sec": 21315, + "stage_training_flops": 3.79596155387904e+16, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.5329203755535565e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/model_000007.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/model_000007.pt new file mode 100644 index 0000000000000000000000000000000000000000..93188a16a277a2a7ae8c9dfb8219b9f85520cb74 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/model_000007.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3c16eb054f43f54f05624b42a4ec3fbcab666b80d99c3c87ac6ee8b34261a131 +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/optim_000007_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/optim_000007_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..cc9feffc054f9812e3c1890c16ee483d079daf05 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/checkpoints/optim_000007_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c53d48431c7459eae0e562672d69b2dc15a15f91e1243ce120d54ab6b64ba458 +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json new file mode 100644 index 0000000000000000000000000000000000000000..1b97707ee9231149762d0f88f4d8ba80b96831b8 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/config.json @@ -0,0 +1,32 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-authentic", + "data": { + "recipe": "pre1930", + "pre1930_epochs": 5 + }, + "training": { + "num_iterations": -1, + "device_batch_size": 8, + "eval_every": -1, + "chatcore_every": -1, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "pre1930", + "ratio20" + ] + }, + "config_fingerprint": "d3378357cef17359", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/run.json new file mode 100644 index 0000000000000000000000000000000000000000..8ba5b17470a8d0c9e6a7acae2e34be4f07c4e9b1 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-authentic", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": null, + "config_fingerprint": "d3378357cef17359", + "wandb_run_id": "0837926f", + "created_at": 1784577842 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/eval_metrics.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/eval_metrics.json new file mode 100644 index 0000000000000000000000000000000000000000..f4856feb918ed0240238d249b6b784926bf0b743 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/eval_metrics.json @@ -0,0 +1,30 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", + "recipe": "curriculum", + "step": 52, + "val_bpb": 0.8869868096604953, + "min_val_bpb": 0.8464286315951827, + "per_route_bpb": {}, + "per_domain_bpb": {}, + "curriculum_summary": null, + "chatcore": { + "chatcore_metric": -0.006895505470088888, + "chatcore_cat": -0.011492509116814813, + "suite": { + "tasks": [ + "ARC-Easy", + "ARC-Challenge", + "MMLU", + "GSM8K", + "SpellingBee" + ], + "max_generative_problems": 32, + "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." + }, + "ARC-Easy": 0.24494949494949494, + "ARC-Challenge": 0.2354948805460751, + "MMLU": 0.24369747899159663, + "GSM8K": 0.0, + "SpellingBee": 0.0 + } +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/meta_000052.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/meta_000052.json new file mode 100644 index 0000000000000000000000000000000000000000..566d3fd5fd3bd34383abba37c5aa0f6776d725c3 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/meta_000052.json @@ -0,0 +1,161 @@ +{ + "step": 52, + "training_complete": true, + "val_bpb": 0.8869868096604953, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", + "wandb_run_id": "fb8277f3", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c0,baseline", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints", + "tokenizer_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", + "experiment_config": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 200, + "eval_tokens": 20971520, + "chatcore_every": -1, + "chatcore_max_cat": -1, + "chatcore_max_sample": 24, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c0", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C0", + "mode": "flat", + "epochs": 3, + "threshold_default": 90, + "routes": { + "knowledge_qa": { + "count": 18000 + }, + "multiturn_qa": { + "count": 18000 + }, + "reasoning_qa": { + "count": 12000 + }, + "narrative_grounded": { + "count": 12000 + }, + "opinion_qa": { + "count": 10000 + }, + "composition_qa": { + "count": 10000 + }, + "how_to_qa": { + "count": 7000 + }, + "verse_qa": { + "count": 6000 + }, + "narrative_fiction": { + "count": 6000 + }, + "stem_reasoning": { + "count": 4000 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": -1, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c0", + "baseline" + ] + }, + "config_fingerprint": "4ce60ef1ccb8a8c8", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "4ce60ef1ccb8a8c8" + }, + "loop_state": { + "step": 52, + "total_training_time": 2071.3676829338074, + "min_val_bpb": 0.8464286315951827, + "smooth_train_loss": 1.608637469780549, + "mfu": 35.219659061221186, + "tok_per_sec": 21247, + "stage_training_flops": 2.819857154310144e+17, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.557322985542779e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/model_000052.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/model_000052.pt new file mode 100644 index 0000000000000000000000000000000000000000..19d84e58368f8e1ef9b337f84c76056060ec3d5c --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/model_000052.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:70c91228c289d6b4d6928d866710be6bda73f21a0e8c1f2a772cd0cd593e2806 +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/optim_000052_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/optim_000052_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..4dd39a75dd77093bc084924d8e43e762567e83cb --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/checkpoints/optim_000052_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6e93a9ac2c446008d78c0697dcc6b159bdb9504247ae98a271535459cd527493 +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json new file mode 100644 index 0000000000000000000000000000000000000000..c3133e425f32aeabd331a214cf024a2d2e7540e4 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/config.json @@ -0,0 +1,78 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c0", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C0", + "mode": "flat", + "epochs": 3, + "threshold_default": 90, + "routes": { + "knowledge_qa": { + "count": 18000 + }, + "multiturn_qa": { + "count": 18000 + }, + "reasoning_qa": { + "count": 12000 + }, + "narrative_grounded": { + "count": 12000 + }, + "opinion_qa": { + "count": 10000 + }, + "composition_qa": { + "count": 10000 + }, + "how_to_qa": { + "count": 7000 + }, + "verse_qa": { + "count": 6000 + }, + "narrative_fiction": { + "count": 6000 + }, + "stem_reasoning": { + "count": 4000 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": -1, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c0", + "baseline" + ] + }, + "config_fingerprint": "4ce60ef1ccb8a8c8", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/evals/chatcore.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/evals/chatcore.json new file mode 100644 index 0000000000000000000000000000000000000000..4c56dd37bf29d9e77abae24b1e0c752bfd313a88 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/evals/chatcore.json @@ -0,0 +1,27 @@ +{ + "stage": "sft", + "step": 52, + "total_training_flops": 4.557322985542779e+19, + "stage_training_flops": 2.819857154310144e+17, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.557322985542779e+19, + "results": { + "ARC-Easy": 0.24494949494949494, + "ARC-Challenge": 0.2354948805460751, + "MMLU": 0.24369747899159663, + "GSM8K": 0.0, + "SpellingBee": 0.0 + }, + "chatcore_metric": -0.006895505470088888, + "chatcore_suite": { + "tasks": [ + "ARC-Easy", + "ARC-Challenge", + "MMLU", + "GSM8K", + "SpellingBee" + ], + "max_generative_problems": 32, + "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/run.json new file mode 100644 index 0000000000000000000000000000000000000000..d3d9fd3be15bed080d5915db0d0c022f1bf0f058 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0/run.json @@ -0,0 +1,8 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c0", + "wandb_run_id": "fb8277f3", + "created_at": 1786488813, + "recovered_from_checkpoint_step": 52, + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/meta_000003.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/meta_000003.json new file mode 100644 index 0000000000000000000000000000000000000000..45cd36180ac79aef95b828b412607b6a268bf1e4 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/meta_000003.json @@ -0,0 +1,181 @@ +{ + "step": 3, + "training_complete": true, + "val_bpb": 0.8281267785987712, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust", + "wandb_run_id": "d8812a13", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c1,minimalist,lima,robustness,noise", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust", + "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "e315bfb5d50d37456aea0c20a8e8a9109c12f9c7", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 100, + "eval_tokens": 20971520, + "chatcore_every": 100000, + "chatcore_max_cat": -1, + "chatcore_max_sample": 24, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c1-robust", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C1R", + "mode": "flat", + "epochs": 4, + "threshold_default": 97, + "routes": { + "knowledge_qa": { + "count": 600 + }, + "multiturn_qa": { + "count": 500 + }, + "reasoning_qa": { + "count": 400 + }, + "narrative_grounded": { + "count": 400 + }, + "opinion_qa": { + "count": 400 + }, + "composition_qa": { + "count": 400 + }, + "how_to_qa": { + "count": 300 + }, + "stem_reasoning": { + "count": 300 + }, + "verse_qa": { + "count": 300 + }, + "narrative_fiction": { + "count": 200 + } + }, + "calibration_qa": { + "count": 300 + }, + "authentic": { + "count": 1000 + }, + "noise": { + "rate": 0.3 + }, + "robustness": { + "epochs": 2, + "routes": { + "conversation_qa": { + "count": 300 + }, + "unparseable_qa": { + "count": 200 + }, + "era_qa": { + "count": 100 + } + } + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 100, + "chatcore_every": 100000, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c1", + "minimalist", + "lima", + "robustness", + "noise" + ] + }, + "config_fingerprint": "6dd68331f59b8eb4", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "6dd68331f59b8eb4" + }, + "loop_state": { + "step": 3, + "total_training_time": 0.0, + "min_val_bpb": 0.8281267785987712, + "smooth_train_loss": 0.7139463255405425, + "mfu": 35.07294255982036, + "tok_per_sec": 21159, + "stage_training_flops": 1.62684066594816e+16, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.530751254665626e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/model_000003.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/model_000003.pt new file mode 100644 index 0000000000000000000000000000000000000000..26b07ebde9cfdcd2f54a8ad87c542dc298aeca2c --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/model_000003.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a10e1430b0ab2b416762c891acf5c8d0fc987e7272a4086266bd096fd80d6dec +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/optim_000003_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/optim_000003_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..a9850986237c9f0951f2877d96d8901b941c3642 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/checkpoints/optim_000003_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:776dabeeabf2167abc5be17a01b6ab81a4aab9b91fa5eb69a15ad1136522a77b +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json new file mode 100644 index 0000000000000000000000000000000000000000..02a1d75ec94de4f8df93ed5fd9119926db23529d --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/config.json @@ -0,0 +1,98 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c1-robust", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C1R", + "mode": "flat", + "epochs": 4, + "threshold_default": 97, + "routes": { + "knowledge_qa": { + "count": 600 + }, + "multiturn_qa": { + "count": 500 + }, + "reasoning_qa": { + "count": 400 + }, + "narrative_grounded": { + "count": 400 + }, + "opinion_qa": { + "count": 400 + }, + "composition_qa": { + "count": 400 + }, + "how_to_qa": { + "count": 300 + }, + "stem_reasoning": { + "count": 300 + }, + "verse_qa": { + "count": 300 + }, + "narrative_fiction": { + "count": 200 + } + }, + "calibration_qa": { + "count": 300 + }, + "authentic": { + "count": 1000 + }, + "noise": { + "rate": 0.3 + }, + "robustness": { + "epochs": 2, + "routes": { + "conversation_qa": { + "count": 300 + }, + "unparseable_qa": { + "count": 200 + }, + "era_qa": { + "count": 100 + } + } + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 100, + "chatcore_every": 100000, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c1", + "minimalist", + "lima", + "robustness", + "noise" + ] + }, + "config_fingerprint": "6dd68331f59b8eb4", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/run.json new file mode 100644 index 0000000000000000000000000000000000000000..c4ec6de669fa94556cdde3f1adc1a8c5488fada4 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1-robust", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": null, + "branch_parent_step": null, + "config_fingerprint": "6dd68331f59b8eb4", + "wandb_run_id": "d8812a13", + "created_at": 1786564717 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/eval_metrics.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/eval_metrics.json new file mode 100644 index 0000000000000000000000000000000000000000..0f7b7f263f49b9481015744d74dbeabb5c25f9bf --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/eval_metrics.json @@ -0,0 +1,30 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", + "recipe": "curriculum", + "step": 3, + "val_bpb": 0.827509209527302, + "min_val_bpb": 0.827509209527302, + "per_route_bpb": {}, + "per_domain_bpb": {}, + "curriculum_summary": null, + "chatcore": { + "chatcore_metric": -0.009923963742371995, + "chatcore_cat": -0.016539939570619992, + "suite": { + "tasks": [ + "ARC-Easy", + "ARC-Challenge", + "MMLU", + "GSM8K", + "SpellingBee" + ], + "max_generative_problems": 32, + "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." + }, + "ARC-Easy": 0.25252525252525254, + "ARC-Challenge": 0.22866894197952217, + "MMLU": 0.2315909414613303, + "GSM8K": 0.0, + "SpellingBee": 0.0 + } +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/meta_000003.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/meta_000003.json new file mode 100644 index 0000000000000000000000000000000000000000..1bac04363882d6eeaa798629684ca35006f8aa13 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/meta_000003.json @@ -0,0 +1,162 @@ +{ + "step": 3, + "training_complete": true, + "val_bpb": 0.827509209527302, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", + "wandb_run_id": "929f01d4", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c1,minimalist,lima", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints", + "tokenizer_dir": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", + "experiment_config": "/workspace/nanochat-state/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 100, + "eval_tokens": 20971520, + "chatcore_every": 100000, + "chatcore_max_cat": -1, + "chatcore_max_sample": 24, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c1", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C1", + "mode": "flat", + "epochs": 4, + "threshold_default": 97, + "routes": { + "knowledge_qa": { + "count": 600 + }, + "multiturn_qa": { + "count": 500 + }, + "reasoning_qa": { + "count": 400 + }, + "narrative_grounded": { + "count": 400 + }, + "opinion_qa": { + "count": 400 + }, + "composition_qa": { + "count": 400 + }, + "how_to_qa": { + "count": 300 + }, + "stem_reasoning": { + "count": 300 + }, + "verse_qa": { + "count": 300 + }, + "narrative_fiction": { + "count": 200 + } + }, + "calibration_qa": { + "count": 300 + }, + "authentic": { + "count": 1000 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 100, + "chatcore_every": 100000, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c1", + "minimalist", + "lima" + ] + }, + "config_fingerprint": "da4f59f54e9dac3b", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "da4f59f54e9dac3b" + }, + "loop_state": { + "step": 3, + "total_training_time": 0.0, + "min_val_bpb": 0.827509209527302, + "smooth_train_loss": 0.705148291826248, + "mfu": 35.12240192928327, + "tok_per_sec": 21189, + "stage_training_flops": 1.62684066594816e+16, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.530751254665626e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/model_000003.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/model_000003.pt new file mode 100644 index 0000000000000000000000000000000000000000..8fee18401fffc1fed7f35b4efbee8fd9710cfc12 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/model_000003.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d615b5c5f303e2177711c7c151a28953692d76060cac82f38803443c9a62c62a +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/optim_000003_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/optim_000003_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..aec65a9096aca26b77b5648dffe4a17e308c46f3 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/checkpoints/optim_000003_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c69c388a2cb60df3b4c0e0c12951a981f6de5ae19f5b1c1742d0080628a2491 +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json new file mode 100644 index 0000000000000000000000000000000000000000..28d06f8c4549f6378abda73e6d02529dfde31da6 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/config.json @@ -0,0 +1,79 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c1", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C1", + "mode": "flat", + "epochs": 4, + "threshold_default": 97, + "routes": { + "knowledge_qa": { + "count": 600 + }, + "multiturn_qa": { + "count": 500 + }, + "reasoning_qa": { + "count": 400 + }, + "narrative_grounded": { + "count": 400 + }, + "opinion_qa": { + "count": 400 + }, + "composition_qa": { + "count": 400 + }, + "how_to_qa": { + "count": 300 + }, + "stem_reasoning": { + "count": 300 + }, + "verse_qa": { + "count": 300 + }, + "narrative_fiction": { + "count": 200 + } + }, + "calibration_qa": { + "count": 300 + }, + "authentic": { + "count": 1000 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 100, + "chatcore_every": 100000, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c1", + "minimalist", + "lima" + ] + }, + "config_fingerprint": "da4f59f54e9dac3b", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/evals/chatcore.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/evals/chatcore.json new file mode 100644 index 0000000000000000000000000000000000000000..ebc8c1a7e19b73e6e84f6d7ac3f3ee2aeeb85a22 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/evals/chatcore.json @@ -0,0 +1,27 @@ +{ + "stage": "sft", + "step": 3, + "total_training_flops": 4.530751254665626e+19, + "stage_training_flops": 1.62684066594816e+16, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.530751254665626e+19, + "results": { + "ARC-Easy": 0.25252525252525254, + "ARC-Challenge": 0.22866894197952217, + "MMLU": 0.2315909414613303, + "GSM8K": 0.0, + "SpellingBee": 0.0 + }, + "chatcore_metric": -0.009923963742371995, + "chatcore_suite": { + "tasks": [ + "ARC-Easy", + "ARC-Challenge", + "MMLU", + "GSM8K", + "SpellingBee" + ], + "max_generative_problems": 32, + "generative_answer_format": "End your response with #### followed by the final numeric answer (for example: #### 42)." + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/run.json new file mode 100644 index 0000000000000000000000000000000000000000..dcf9681ef472a6272bfd8c19accbd3955f0c3dd5 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1/run.json @@ -0,0 +1,8 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c1", + "wandb_run_id": "929f01d4", + "created_at": 1786489421, + "recovered_from_checkpoint_step": 3, + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/meta_000040.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/meta_000040.json new file mode 100644 index 0000000000000000000000000000000000000000..23c0d9ea09b5fc41699996348d9008c104f20ad0 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/meta_000040.json @@ -0,0 +1,163 @@ +{ + "step": 40, + "training_complete": true, + "val_bpb": 0.8800241308853763, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2", + "wandb_run_id": "69545234", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c2,reasoning-forward", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2", + "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 200, + "eval_tokens": 20971520, + "chatcore_every": 100000, + "chatcore_max_cat": -1, + "chatcore_max_sample": 32, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c2", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C2", + "mode": "flat", + "epochs": 3, + "threshold_default": 90, + "routes": { + "reasoning_qa": { + "count": 16000 + }, + "how_to_qa": { + "count": null + }, + "stem_reasoning": { + "count": null, + "threshold": 80 + }, + "knowledge_qa": { + "count": 10000 + }, + "multiturn_qa": { + "count": 10000 + }, + "narrative_grounded": { + "count": 6000 + }, + "opinion_qa": { + "count": 6000 + }, + "composition_qa": { + "count": 6000 + }, + "verse_qa": { + "count": 3000 + }, + "narrative_fiction": { + "count": 3000 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c2", + "reasoning-forward" + ] + }, + "config_fingerprint": "74c02b2162b87713", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "74c02b2162b87713" + }, + "loop_state": { + "step": 40, + "total_training_time": 1483.5767045021057, + "min_val_bpb": 0.8465480228052693, + "smooth_train_loss": 1.6970454443686784, + "mfu": 35.06659423129714, + "tok_per_sec": 21155, + "stage_training_flops": 2.16912088793088e+17, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.550815622878986e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/model_000040.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/model_000040.pt new file mode 100644 index 0000000000000000000000000000000000000000..2118defd9129f117ad18b1de1e7f1dc5b786da1a --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/model_000040.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f9ef0df02226735f57dc26525d46aa2be920b67bdcc6244f7a1475f10b6b3e00 +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/optim_000040_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/optim_000040_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..ba948db1373686190827e009d1fd75b0f14fa709 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/checkpoints/optim_000040_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c0c41f8550f90563488e72cb3afd0244964901c668f2e4b552979d23d0a21daa +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json new file mode 100644 index 0000000000000000000000000000000000000000..3a022d15ffd4cfa504d7f3ca9fd2567ae12f62cc --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/config.json @@ -0,0 +1,80 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c2", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C2", + "mode": "flat", + "epochs": 3, + "threshold_default": 90, + "routes": { + "reasoning_qa": { + "count": 16000 + }, + "how_to_qa": { + "count": null + }, + "stem_reasoning": { + "count": null, + "threshold": 80 + }, + "knowledge_qa": { + "count": 10000 + }, + "multiturn_qa": { + "count": 10000 + }, + "narrative_grounded": { + "count": 6000 + }, + "opinion_qa": { + "count": 6000 + }, + "composition_qa": { + "count": 6000 + }, + "verse_qa": { + "count": 3000 + }, + "narrative_fiction": { + "count": 3000 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c2", + "reasoning-forward" + ] + }, + "config_fingerprint": "74c02b2162b87713", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/run.json new file mode 100644 index 0000000000000000000000000000000000000000..e39a03236040cea60a7ad4da9efec06c7a25d013 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c2", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352, + "branch_parent_step": null, + "config_fingerprint": "74c02b2162b87713", + "wandb_run_id": "69545234", + "created_at": 1786490713 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/meta_000082.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/meta_000082.json new file mode 100644 index 0000000000000000000000000000000000000000..736c1e3f483cb3c1f9ac2f22cd214ff43a99707a --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/meta_000082.json @@ -0,0 +1,151 @@ +{ + "step": 82, + "training_complete": true, + "val_bpb": 0.7636363678276777, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3", + "wandb_run_id": "1720db37", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c3,scale-max,staged", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3", + "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 400, + "eval_tokens": 20971520, + "chatcore_every": 100000, + "chatcore_max_cat": -1, + "chatcore_max_sample": 32, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c3", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C3", + "mode": "staged", + "threshold_default": 80, + "stages": [ + { + "routes": [ + "knowledge_qa" + ], + "authentic": "single" + }, + { + "routes": [ + "reasoning_qa", + "stem_reasoning", + "how_to_qa", + "opinion_qa", + "composition_qa", + "verse_qa" + ], + "calibration_qa": true + }, + { + "routes": [ + "multiturn_qa", + "narrative_grounded", + "narrative_fiction" + ], + "authentic": "multi" + } + ] + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 400, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c3", + "scale-max", + "staged" + ] + }, + "config_fingerprint": "a86d67a1560d2efa", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "a86d67a1560d2efa" + }, + "loop_state": { + "step": 82, + "total_training_time": 3557.9933972358704, + "min_val_bpb": 0.7636363678276777, + "smooth_train_loss": 2.003534360747555, + "mfu": 35.111241842899126, + "tok_per_sec": 21182, + "stage_training_flops": 4.446697820258304e+17, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.5735913922022605e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/model_000082.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/model_000082.pt new file mode 100644 index 0000000000000000000000000000000000000000..23ef1542c884657f8f827f89460ab8b640e4ea57 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/model_000082.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:559f68d8f9c396481864e571c147c339a6858e492a5edfc1810e7c9efae149dd +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/optim_000082_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/optim_000082_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..89a93a505d07b7d1966fb19879690d3554fa5e30 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/checkpoints/optim_000082_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1a05a322ddaf93293b6c9fdd335cdb48104190ac2368a050f41a37fa6c49c4b7 +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json new file mode 100644 index 0000000000000000000000000000000000000000..6c6dc403e5ed750bd2664a1c8849e491943208af --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/config.json @@ -0,0 +1,68 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c3", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C3", + "mode": "staged", + "threshold_default": 80, + "stages": [ + { + "routes": [ + "knowledge_qa" + ], + "authentic": "single" + }, + { + "routes": [ + "reasoning_qa", + "stem_reasoning", + "how_to_qa", + "opinion_qa", + "composition_qa", + "verse_qa" + ], + "calibration_qa": true + }, + { + "routes": [ + "multiturn_qa", + "narrative_grounded", + "narrative_fiction" + ], + "authentic": "multi" + } + ] + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 400, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c3", + "scale-max", + "staged" + ] + }, + "config_fingerprint": "a86d67a1560d2efa", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/run.json new file mode 100644 index 0000000000000000000000000000000000000000..8f70cb389b3fab0520d0864478ec47bfccce2a2e --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c3", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352, + "branch_parent_step": null, + "config_fingerprint": "a86d67a1560d2efa", + "wandb_run_id": "1720db37", + "created_at": 1786499635 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/meta_000030.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/meta_000030.json new file mode 100644 index 0000000000000000000000000000000000000000..bf9f8f9b84ec407ddbd87e7a206898f0129ae4b9 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/meta_000030.json @@ -0,0 +1,163 @@ +{ + "step": 30, + "training_complete": true, + "val_bpb": 0.8584271178670264, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4", + "wandb_run_id": "4dd1e026", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c4,token-balanced", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4", + "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 200, + "eval_tokens": 20971520, + "chatcore_every": 100000, + "chatcore_max_cat": -1, + "chatcore_max_sample": 32, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c4", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C4", + "mode": "flat", + "epochs": 3, + "threshold_default": 90, + "_note": "counts pre-derived to ~4M chars (~1M tokens) per route (token-balanced)", + "routes": { + "knowledge_qa": { + "count": 13150 + }, + "opinion_qa": { + "count": 10950 + }, + "how_to_qa": { + "count": 7450 + }, + "verse_qa": { + "count": 5530 + }, + "reasoning_qa": { + "count": 4880 + }, + "multiturn_qa": { + "count": 4820 + }, + "narrative_grounded": { + "count": 4460 + }, + "composition_qa": { + "count": 4330 + }, + "stem_reasoning": { + "count": 3800 + }, + "narrative_fiction": { + "count": 2840 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c4", + "token-balanced" + ] + }, + "config_fingerprint": "331c13b7908a3efc", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "331c13b7908a3efc" + }, + "loop_state": { + "step": 30, + "total_training_time": 988.5635514259338, + "min_val_bpb": 0.8463398712741395, + "smooth_train_loss": 1.813480427055826, + "mfu": 35.10521097749975, + "tok_per_sec": 21178, + "stage_training_flops": 1.62684066594816e+17, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.545392820659159e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/model_000030.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/model_000030.pt new file mode 100644 index 0000000000000000000000000000000000000000..b3146523722931d756a499498c0e59ac90b6a1e5 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/model_000030.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:05c335bd1d3b1d0a09dcb9498c6c4bd6c0ba1f1170b979f0ccc87348dbc61fa5 +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/optim_000030_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/optim_000030_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..bc6ce0d089955e19eb0606a419ae40dd003b2d23 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/checkpoints/optim_000030_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4ab2b8e11a75e45dbb12419ce73d08f62d91dd5e3fe092984f6eb26f020f0ceb +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json new file mode 100644 index 0000000000000000000000000000000000000000..29644571da19f093a9f0a829984dc2b4c7844886 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/config.json @@ -0,0 +1,80 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c4", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C4", + "mode": "flat", + "epochs": 3, + "threshold_default": 90, + "_note": "counts pre-derived to ~4M chars (~1M tokens) per route (token-balanced)", + "routes": { + "knowledge_qa": { + "count": 13150 + }, + "opinion_qa": { + "count": 10950 + }, + "how_to_qa": { + "count": 7450 + }, + "verse_qa": { + "count": 5530 + }, + "reasoning_qa": { + "count": 4880 + }, + "multiturn_qa": { + "count": 4820 + }, + "narrative_grounded": { + "count": 4460 + }, + "composition_qa": { + "count": 4330 + }, + "stem_reasoning": { + "count": 3800 + }, + "narrative_fiction": { + "count": 2840 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c4", + "token-balanced" + ] + }, + "config_fingerprint": "331c13b7908a3efc", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/run.json new file mode 100644 index 0000000000000000000000000000000000000000..9d9d1637f4b76fc267f36c1139f47a8ef1553323 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c4", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352, + "branch_parent_step": null, + "config_fingerprint": "331c13b7908a3efc", + "wandb_run_id": "4dd1e026", + "created_at": 1786505289 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/meta_000052.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/meta_000052.json new file mode 100644 index 0000000000000000000000000000000000000000..97e76add1e4077375fd32822a8e4ab36e854385f --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/meta_000052.json @@ -0,0 +1,163 @@ +{ + "step": 52, + "training_complete": true, + "val_bpb": 0.8903500530127588, + "model_config": { + "sequence_len": 4096, + "vocab_size": 32768, + "n_layer": 24, + "n_head": 12, + "n_kv_head": 12, + "n_embd": 1536, + "window_pattern": "SSSL" + }, + "user_config": { + "run": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5", + "wandb_run_id": "17210657", + "wandb_group": "think-d12", + "wandb_tags": "sft,curriculum,c5,domain-rebalanced", + "device_type": "", + "model_tag": null, + "model_step": null, + "base_checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/base_checkpoints", + "base_step": 8352, + "checkpoint_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints", + "tokenizer_dir": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer", + "resume_from_step": null, + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5", + "experiment_config": "/content/nanochat_cache/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json", + "parent_cumulative_flops": 4.5291244139996774e+19, + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "git_commit_sha": "59ab2594ad6fe85e4294aada37a6f0a0284eebda", + "load_optimizer": 0, + "num_iterations": -1, + "max_seq_len": null, + "device_batch_size": 8, + "total_batch_size": null, + "embedding_lr": null, + "unembedding_lr": null, + "matrix_lr": null, + "init_lr_frac": 0.8, + "warmup_ratio": 0.03, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": 200, + "eval_tokens": 20971520, + "chatcore_every": 100000, + "chatcore_max_cat": -1, + "chatcore_max_sample": 32, + "save_every": -1, + "recipe": "curriculum", + "curriculum_config": "", + "pre1930_epochs": 5, + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "authentic_epochs": 0, + "knowledge_qa_epochs": 0, + "multiturn_qa_epochs": 0, + "reasoning_qa_epochs": 0, + "stem_reasoning_epochs": 0, + "narrative_grounded_epochs": 0, + "narrative_fiction_epochs": 0, + "opinion_qa_epochs": 0, + "how_to_qa_epochs": 0, + "verse_qa_epochs": 0, + "composition_qa_epochs": 0, + "resolved_experiment_config": { + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c5", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C5", + "mode": "domain_rebalanced", + "epochs": 3, + "threshold_default": 90, + "_note": "same route totals as C0; book_category flattened within each route", + "routes": { + "knowledge_qa": { + "count": 18000 + }, + "multiturn_qa": { + "count": 18000 + }, + "reasoning_qa": { + "count": 12000 + }, + "narrative_grounded": { + "count": 12000 + }, + "opinion_qa": { + "count": 10000 + }, + "composition_qa": { + "count": 10000 + }, + "how_to_qa": { + "count": 7000 + }, + "verse_qa": { + "count": 6000 + }, + "narrative_fiction": { + "count": 6000 + }, + "stem_reasoning": { + "count": 4000 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c5", + "domain-rebalanced" + ] + }, + "config_fingerprint": "b1bbd0a977780ed3", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5" + }, + "stage": "sft", + "base_experiment_id": null, + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "b1bbd0a977780ed3" + }, + "loop_state": { + "step": 52, + "total_training_time": 2075.6576771736145, + "min_val_bpb": 0.8464341895781785, + "smooth_train_loss": 1.6294214116602808, + "mfu": 35.10245957312791, + "tok_per_sec": 21177, + "stage_training_flops": 2.819857154310144e+17, + "inherited_parent_flops": 4.5291244139996774e+19, + "cumulative_pipeline_training_flops": 4.557322985542779e+19 + } +} \ No newline at end of file diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/model_000052.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/model_000052.pt new file mode 100644 index 0000000000000000000000000000000000000000..ff34b4e7a5c19b05a2e558199e3f676354a2b8aa --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/model_000052.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ec260000ca50247d558428abdf8630b5dd5f568e0b635ba803646617a82fee1 +size 4227936073 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/optim_000052_rank0.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/optim_000052_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..27850ea21de6c5714e3291ca359ad80d8aceff09 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/checkpoints/optim_000052_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5df00f34d178cf428a8fdfcb42733d31195ea174cf37d3b45e6106b13360d476 +size 5739601781 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json new file mode 100644 index 0000000000000000000000000000000000000000..eedd423d3c5c599507d4fcf9ecb94e300b0963b9 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/config.json @@ -0,0 +1,80 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "pre1930-curriculum-c5", + "data": { + "recipe": "curriculum", + "curriculum": { + "name": "C5", + "mode": "domain_rebalanced", + "epochs": 3, + "threshold_default": 90, + "_note": "same route totals as C0; book_category flattened within each route", + "routes": { + "knowledge_qa": { + "count": 18000 + }, + "multiturn_qa": { + "count": 18000 + }, + "reasoning_qa": { + "count": 12000 + }, + "narrative_grounded": { + "count": 12000 + }, + "opinion_qa": { + "count": 10000 + }, + "composition_qa": { + "count": 10000 + }, + "how_to_qa": { + "count": 7000 + }, + "verse_qa": { + "count": 6000 + }, + "narrative_fiction": { + "count": 6000 + }, + "stem_reasoning": { + "count": 4000 + } + }, + "calibration_qa": { + "count": null + }, + "authentic": { + "count": 12257 + } + } + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "device_batch_size": 8, + "warmup_ratio": 0.03, + "eval_every": 200, + "chatcore_every": 100000, + "chatcore_max_sample": 32, + "save_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "think-d12", + "tags": [ + "sft", + "curriculum", + "c5", + "domain-rebalanced" + ] + }, + "config_fingerprint": "b1bbd0a977780ed3", + "artifact_path": "experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5" +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/run.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/run.json new file mode 100644 index 0000000000000000000000000000000000000000..c92e891658640a2ad96ea69840d3893d2def0a1f --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/sft/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1-pre1930-curriculum-c5", + "stage": "sft", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_checkpoint_step": 8352, + "branch_parent_step": null, + "config_fingerprint": "b1bbd0a977780ed3", + "wandb_run_id": "17210657", + "created_at": 1786508150 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/summary.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..fa27f8b246994f1fd96b1fead4b04e441cad27cb --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/summary.json @@ -0,0 +1,92 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "stage": "base", + "base_experiment_id": "clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset-clean-1930s", + "dataset_revision": "main", + "step": 8352, + "depth": 24, + "target_param_data_ratio": 12, + "training_tokens": 8757706752, + "final_sampled_val_bpb": 0.9204610080723857, + "minimum_sampled_val_bpb": 0.9204610080723857, + "full_val_bpb": null, + "core_metric": 0.13542864491271994, + "centered_results": { + "hellaswag_zeroshot": 0.09818760553995769, + "jeopardy": 0.0037789323832839727, + "bigbench_qa_wikidata": 0.2516116201877594, + "arc_easy": 0.22334452470143637, + "arc_challenge": -0.013651887575785318, + "copa": 0.2200000286102295, + "commonsense_qa": 0.13390662521123883, + "piqa": 0.16104459762573242, + "openbook_qa": 0.010666688283284506, + "lambada_openai": 0.35338637232780457, + "hellaswag": 0.105490247408549, + "winograd": 0.34065937995910645, + "winogrande": -0.011839032173156738, + "bigbench_dyck_languages": 0.12700000405311584, + "agi_eval_lsat_ar": 0.07608695328235625, + "bigbench_cs_algorithms": 0.38333332538604736, + "bigbench_operators": 0.11428572237491608, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.114096499979496, + "coqa": 0.16610296070575714, + "boolq": -0.060679241230613294, + "bigbench_language_identification": 0.1826182610393226 + }, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is the city of Paris, which is the capital of France. The capital of France" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is Au, and the symbol of silver is Ag. The symbol of gold is Au" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If yesterday was Friday, then tomorrow will be Saturday. If yesterday was" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is cold.\n\nThe opposite of cold is hot.\n\nThe opposite of hot is cold.\n\n" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: Mercury, Venus, Earth, Mars, Jupiter, Saturn, Uranus," + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is a dark brown, with a tinge of red. It is a very pretty color" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is 13, and the equation is 13x + 3 = 13" + } + ], + "unconditioned_samples": [ + "<|bos|>PREFACE.\n\nTHIS essay was contributed to the \"Psychological Bulletin,\" of the University of Chicago, during the past summer; the temporary arrangements for its printing prevented its meeting the demand for it. Fair play in the matter of results in the revision of Shakespeare's plays has not been given in this field in this company, Dunning, Uzziel, and others engaging to contribute whatever they have to say on the revision and construction of even a very few plays, and the magazine enterprise to furnish more plays of a classic nature, although comprising many dramas of merit and of the utmost possible variety, is by no means of sufficient scope and interest", + "<|bos|>\n\nARTON LIBRARY OF THE DEZA SUPERINTENDENTS AND BOOKKEEPERS\n\nThe following Outline of Business\n\nthe State or Territory in which\n\nFoss work is to be carried on.\n\nPLAN OF BUSINESS Department 1. Traveling\n\n2. Records, coupons, checks, mercantile accounts, vouchers, &c.\n\n3. General office\n\n4. Subsidiary agencies. Persons dealing with department men.\n\nPlanwork.\n\nPlan.\n\n1. Plans for the method of review of machinery, etc. Uncertainty.\n\n2. Plans for securing uniformity of practice over a large field", + "<|bos|>!\n\nIt's the coss is mighty wonderous weak Thefe minutes yhere hain marry teef The tryfector\n\nOf my fandd in Life who ought to hast The pist o' a President\n\nApril 10. J,\"\n\nAmong the Arms in cison\n\nThe Arms of the President *The Arms of Washington\n\nOn October 5, (?) in another and more noticeable storm-swept plagiarism, the President bore the baptismal emblem of the United States; on the way to Washington, January 1, 1799, occurred the cancel that has made possible the interpolations of the following", + "<|bos|> ministered unto Him with great joy.\"\n\nXX. 1. CHIRIT OF THE BLESSED.\n\nThe full contrast to this joy is the hunger that followed when He left His throne in heaven, ascended, and took the place of the \"Name\" lifted up on Calvary.\n\n2. The Fulfilment of the Tabernacle: In our imagination let 50 Enter: first, the Church universal: prayer, glorification, rejoicing, nothing but an eternal army and the eternal joy which forms one side of this-and well! for consequently this is that infant Church which a voice descended, and the Redeemer Sunday broke through the wall that", + "<|bos|>eed\n\nNew York paid\n\nEach separate $1.25 and i piece by akely ried in Watts averaged 75 cents.\n\nRetail price 50 cents-No longer find it worth while to order second-hand books. Already considerable order great publishing interests are sending agents to England or Scotland to our mmission depot expressly to collect the $1.25 per set, on which income is made over to us. The influens can gain fifteen per cent, and this can be taken care of at once as it is here almost never debited. Its great advantages that has the most of A. L. Burlingame's", + "<|bos|>PREFACE\n\nWHEN PERSHAD was about to issue and was being attacked by the fanatical cuttlefish of Northern India, some weeks before at least the Calcutta of 1862 stood face to face with a panic-crazed gale, the doctors began one day on their platform to advise the world how they were going to appease the spirit of the Fish-god. The Elders and the Lady-Seconds criticized emphatically the pros and cons, and the whole city stared.\n\nOne man said: \"I am Oriental; the people of this English country do not believe in all this agnostic nonsense, and the skies when they are", + "<|bos|>iman have published,\n\nTo be completed in about 100 Monthly Parts, each illustrated by a Steel Engraving, a Work consisting of a Library of Original 180 Engravings, 3 vols. 8vo. \u00a35. 5s. strongly bound in cloth, as a Work for convenient\n\nSale by STEEL Engraved by W. L. STEENERSON, each neatly illustrated by a Steel Engraving; the Plates are in a very high state of\n\nPerfection, and warranted to give a correct Image of each Novel.\n\nThe Work is published in Monthly Parts to contain Historical and Biographical Sketches of the principal", + "<|bos|>MYSTER LIBRARY.\n\n3 2044 009 603 166 fidh\n\nQUE\n\n7808, W" + ], + "training_time_seconds": 18494.615287542343, + "stage_training_flops": 4.5291244139996774e+19, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 4.5291244139996774e+19, + "config_fingerprint": "166e7e695c1b536a", + "git_commit_sha": "60aa7005265901858554bb7ca98cd788de721e4a", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/e69c1e59", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1", + "dataset_fingerprint": "0db4605cfe3a7eac", + "tokenizer_fingerprint": "21e99d5cdeeaa660", + "unique_train_tokens": 0 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/experiment_tokenizer.json b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..abb9ccfdfc6557194bfe6a3551bcdd320b777e78 --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "clean1930s-d24-r12-ctx4096-fulltok-v1", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset-clean-1930s", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 200, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 1000000000000, + "doc_cap": 1000000000, + "vocab_size": 32768 + }, + "created_at": 1784129374 +} diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/token_bytes.pt b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..737ab9ff9eafdbd5bfa971d0390b520b87ebb55a --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bc779ae25dfa6f35146f7b9991fa3bab9f2a82a89a4dd92fbd4a57655680e7e2 +size 132649 diff --git a/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/tokenizer.pkl b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..34650d2ed06bbfb645ad394f823340b08c7af1ac --- /dev/null +++ b/experiments/clean1930s-d24-r12-ctx4096-sssl-fulltok-v1/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:155d20e57ea2203cab333207e97b2bec8c0d224678bfce2f019cff3a8ddd940f +size 410542 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_000500.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..cee5997c220788b7c64ac300d4ccaa033751f32f --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_000500.json @@ -0,0 +1,125 @@ +{ + "step": 500, + "experiment_id": "climbmix-d12-1epoch-25shards", + "val_bpb": 1.0398997071863882, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "wandb_group": "climbmix-d12", + "wandb_tags": "climbmix,d12,one-epoch,25-shards", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 2520, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", + "experiment_id": "climbmix-d12-1epoch-25shards", + "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "climbmix-d12-1epoch-25shards", + "experiment": { + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "storage": { + "hf_model_repo": "jbduran/think-nanochat-d12", + "path": "experiments/climbmix-d12-1epoch-25shards" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": [ + "climbmix", + "d12", + "one-epoch", + "25-shards" + ] + } + } + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.0398997071863882, + "smooth_train_loss": 3.4449955375416246, + "total_training_time": 1329.5420286655426 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001000.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..d76553c2ec68d530b6e03e3014a490340e71a635 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001000.json @@ -0,0 +1,125 @@ +{ + "step": 1000, + "experiment_id": "climbmix-d12-1epoch-25shards", + "val_bpb": 0.9820657988478136, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "wandb_group": "climbmix-d12", + "wandb_tags": "climbmix,d12,one-epoch,25-shards", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 2520, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", + "experiment_id": "climbmix-d12-1epoch-25shards", + "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "climbmix-d12-1epoch-25shards", + "experiment": { + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "storage": { + "hf_model_repo": "jbduran/think-nanochat-d12", + "path": "experiments/climbmix-d12-1epoch-25shards" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": [ + "climbmix", + "d12", + "one-epoch", + "25-shards" + ] + } + } + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 0.9820657988478136, + "smooth_train_loss": 3.200258021225873, + "total_training_time": 2687.6018402576447 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001500.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..f009fe6c39fbc093881dc98f3c16d984c8097cc8 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_001500.json @@ -0,0 +1,125 @@ +{ + "step": 1500, + "experiment_id": "climbmix-d12-1epoch-25shards", + "val_bpb": 0.9363922843394795, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "wandb_group": "climbmix-d12", + "wandb_tags": "climbmix,d12,one-epoch,25-shards", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 2520, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", + "experiment_id": "climbmix-d12-1epoch-25shards", + "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "climbmix-d12-1epoch-25shards", + "experiment": { + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "storage": { + "hf_model_repo": "jbduran/think-nanochat-d12", + "path": "experiments/climbmix-d12-1epoch-25shards" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": [ + "climbmix", + "d12", + "one-epoch", + "25-shards" + ] + } + } + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 0.9363922843394795, + "smooth_train_loss": 3.0161736783997206, + "total_training_time": 4045.9229834079742 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002000.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..162f7e88e0269237bca53ef8b693b15faac20b3d --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002000.json @@ -0,0 +1,125 @@ +{ + "step": 2000, + "experiment_id": "climbmix-d12-1epoch-25shards", + "val_bpb": 0.9025021880109514, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "wandb_group": "climbmix-d12", + "wandb_tags": "climbmix,d12,one-epoch,25-shards", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 2520, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", + "experiment_id": "climbmix-d12-1epoch-25shards", + "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "climbmix-d12-1epoch-25shards", + "experiment": { + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "storage": { + "hf_model_repo": "jbduran/think-nanochat-d12", + "path": "experiments/climbmix-d12-1epoch-25shards" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": [ + "climbmix", + "d12", + "one-epoch", + "25-shards" + ] + } + } + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 0.9025021880109514, + "smooth_train_loss": 2.928742685399829, + "total_training_time": 5404.63369512558 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002500.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002500.json new file mode 100644 index 0000000000000000000000000000000000000000..c50d8c79a02fada4d83a52c16e2fa36c21604a07 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002500.json @@ -0,0 +1,125 @@ +{ + "step": 2500, + "experiment_id": "climbmix-d12-1epoch-25shards", + "val_bpb": 0.8791912820900084, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "wandb_group": "climbmix-d12", + "wandb_tags": "climbmix,d12,one-epoch,25-shards", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 2520, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", + "experiment_id": "climbmix-d12-1epoch-25shards", + "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "climbmix-d12-1epoch-25shards", + "experiment": { + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "storage": { + "hf_model_repo": "jbduran/think-nanochat-d12", + "path": "experiments/climbmix-d12-1epoch-25shards" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": [ + "climbmix", + "d12", + "one-epoch", + "25-shards" + ] + } + } + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 13, + "pos": 10792769, + "epoch": 1, + "pq_idx": 13, + "rg_idx": 10792769 + }, + "loop_state": { + "min_val_bpb": 0.8791912820900084, + "smooth_train_loss": 2.873548251177944, + "total_training_time": 6763.973432302475 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002520.json b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002520.json new file mode 100644 index 0000000000000000000000000000000000000000..7892b69cc8482988c3ae57a1ed1fe8162e7b4494 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/meta_002520.json @@ -0,0 +1,125 @@ +{ + "step": 2520, + "experiment_id": "climbmix-d12-1epoch-25shards", + "val_bpb": 0.8786508624512739, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "wandb_group": "climbmix-d12", + "wandb_tags": "climbmix,d12,one-epoch,25-shards", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": 2520, + "target_flops": -1.0, + "target_param_data_ratio": -1.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/base_checkpoints", + "experiment_id": "climbmix-d12-1epoch-25shards", + "experiment_config": "/content/nanochat_cache/experiments/climbmix-d12-1epoch-25shards/config.json", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "climbmix-d12-1epoch-25shards", + "experiment": { + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "storage": { + "hf_model_repo": "jbduran/think-nanochat-d12", + "path": "experiments/climbmix-d12-1epoch-25shards" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": [ + "climbmix", + "d12", + "one-epoch", + "25-shards" + ] + } + } + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 0, + "pos": 73089, + "epoch": 2, + "pq_idx": 0, + "rg_idx": 73089 + }, + "loop_state": { + "min_val_bpb": 0.8786508624512739, + "smooth_train_loss": 2.898973586577961, + "total_training_time": 6818.212191104889 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_000500.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..8b4308c61e02831752bfe146b268984a5ac96ae7 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f7f423d2e924189fcdee1dcf38d7bb7f412de8483b75a83ee20c3c43ef9008d6 +size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001000.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..b86444a6a1a21d37e802c1b9f1a3db7ffd30c35a --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c7860a7e2b7eeea2d597a78eca2e6249523c78aa393059923a7c904c89a78fc8 +size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001500.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..6191ba9da5c9378c4cc43d1dd47edb693337a8f6 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:21f3a4a8eab9fda9eca2832714157828c929b991bf7494e663b2c18128a2905b +size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002000.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..27fc1d2f7b2b41a545767163958c587c3030839c --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc3ba8f7f81c33286862180d8751e8c5deb9cdc030056d79ef8cf26034dede67 +size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002500.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002500.pt new file mode 100644 index 0000000000000000000000000000000000000000..0f9286cd2601a25bd7ba8372af5b9e4a833b6e06 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:291599c9d394e9b6fed99daa0fcffc694946155e441e5cb698d2c5a196a708d6 +size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002520.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002520.pt new file mode 100644 index 0000000000000000000000000000000000000000..5d5894d1fa0006feff90c939534c63d095f314ab --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/model_002520.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fcb5db6082a7323774d8d064f9bce1cc8f8fcfd0618a855c880c4e8ca20e7ab2 +size 792761690 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_000500_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..9ea835127ef2d0b6a2187a6548f5ba35cc57cfff --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:121dc7fb80dd11ee9e3d9057cc91f2705b2395ec453ee1297675cb4023e8f094 +size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001000_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..7cb264f8b5f73b95a14a009f8a95a9f26c015b52 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:184fcc2ff50f7c1504d124edea40ff56dc60cda08bcd0c4ac979c6e8ff479de1 +size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001500_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..6abdd2b9a3af65234e29029223a58b1b1db1c00b --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e05e4efc73a6107d2fdf741f51145f57c3bb4ad138cc65ee8f429592aa12b7b9 +size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002000_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c9551719ebc7009e6cecb7b383e0a51df3cc3b9 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:fdfcf95d93d06395de0fe12cce389019ba3c7c98c7472ef3c4ecdfc74b78dea3 +size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002500_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..dbfea86af94ddd8c5827c7ce1b991e07aef2ad91 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:13a6c3762c33d08005feb0b07e66c6b24381738d20bee44ef2b5378d802530ff +size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002520_rank0.pt b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002520_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..fbca8358dd58220013a920ad4ac26dca9f0a0940 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/base_checkpoints/optim_002520_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:eac3b8631753867c67316583d315c42259fb50914e17a2bbbe84dcb3bf3162cf +size 1246165357 diff --git a/experiments/climbmix-d12-1epoch-25shards/config.json b/experiments/climbmix-d12-1epoch-25shards/config.json new file mode 100644 index 0000000000000000000000000000000000000000..71d511d1ab568741433db485d0fbe874a7f269a2 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/config.json @@ -0,0 +1,50 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 1321205760, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_tokens": 1321205760, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "climbmix-d12-1epoch-25shards", + "group": "climbmix-d12", + "tags": ["climbmix", "d12", "one-epoch", "25-shards"] + } +} diff --git a/experiments/climbmix-d12-1epoch-25shards/evals/core.json b/experiments/climbmix-d12-1epoch-25shards/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..f43d6b4aeb759b2d9b6ad355616199e0b76c931e --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/evals/core.json @@ -0,0 +1,54 @@ +{ + "model": "base_model (step 2520)", + "step": 2520, + "bpb": {}, + "core_metric": 0.1479302855639536, + "core_results": { + "hellaswag_zeroshot": 0.35480979084968567, + "jeopardy": 0.009447330608963966, + "bigbench_qa_wikidata": 0.26726046204566956, + "arc_easy": 0.563973069190979, + "arc_challenge": 0.26365187764167786, + "copa": 0.550000011920929, + "commonsense_qa": 0.3628173768520355, + "piqa": 0.6599564552307129, + "openbook_qa": 0.30400002002716064, + "lambada_openai": 0.31554433703422546, + "hellaswag": 0.34833696484565735, + "winograd": 0.5494505763053894, + "winogrande": 0.5106551051139832, + "bigbench_dyck_languages": 0.0650000050663948, + "agi_eval_lsat_ar": 0.2869565188884735, + "bigbench_cs_algorithms": 0.426515132188797, + "bigbench_operators": 0.10476190596818924, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.14692525565624237, + "coqa": 0.17825378477573395, + "boolq": 0.593883752822876, + "bigbench_language_identification": 0.2522999942302704 + }, + "centered_results": { + "hellaswag_zeroshot": 0.1397463877995809, + "jeopardy": 0.009447330608963966, + "bigbench_qa_wikidata": 0.26726046204566956, + "arc_easy": 0.41863075892130536, + "arc_challenge": 0.01820250352223714, + "copa": 0.10000002384185791, + "commonsense_qa": 0.20352172106504438, + "piqa": 0.3199129104614258, + "openbook_qa": 0.07200002670288086, + "lambada_openai": 0.31554433703422546, + "hellaswag": 0.13111595312754312, + "winograd": 0.09890115261077881, + "winogrande": 0.02131021022796631, + "bigbench_dyck_languages": 0.0650000050663948, + "agi_eval_lsat_ar": 0.10869564861059187, + "bigbench_cs_algorithms": 0.426515132188797, + "bigbench_operators": 0.10476190596818924, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.14692525565624237, + "coqa": 0.17825378477573395, + "boolq": -0.06872696625558952, + "bigbench_language_identification": 0.17744773842714012 + } +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/evals/val_bpb.json b/experiments/climbmix-d12-1epoch-25shards/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..997556146e0dd714c7a659dc441b65ac486cfa8e --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/evals/val_bpb.json @@ -0,0 +1,10 @@ +{ + "model": "base_model (step 2520)", + "step": 2520, + "bpb": { + "val": 0.8792611250626658 + }, + "core_metric": null, + "core_results": null, + "centered_results": null +} \ No newline at end of file diff --git a/experiments/climbmix-d12-1epoch-25shards/run.json b/experiments/climbmix-d12-1epoch-25shards/run.json new file mode 100644 index 0000000000000000000000000000000000000000..07ce2daa96cfbab3d52418d1bd373cbd82662bdf --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/run.json @@ -0,0 +1,5 @@ +{ + "experiment_id": "climbmix-d12-1epoch-25shards", + "wandb_run_id": "2c3bf17b", + "created_at": 1781142948 +} diff --git a/experiments/climbmix-d12-1epoch-25shards/summary.json b/experiments/climbmix-d12-1epoch-25shards/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..070a73bcf0e42cdac605b07d7597d026dbd14301 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/summary.json @@ -0,0 +1,44 @@ +{ + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": "karpathy/climbmix-400b-shuffle", + "dataset_revision": "main", + "step": 2520, + "depth": 12, + "target_param_data_ratio": null, + "training_tokens": 1321205760, + "final_sampled_val_bpb": 0.8786508624512739, + "minimum_sampled_val_bpb": 0.8786508624512739, + "full_val_bpb": 0.8792611250626658, + "core_metric": 0.1479302855639536, + "centered_results": { + "hellaswag_zeroshot": 0.1397463877995809, + "jeopardy": 0.009447330608963966, + "bigbench_qa_wikidata": 0.26726046204566956, + "arc_easy": 0.41863075892130536, + "arc_challenge": 0.01820250352223714, + "copa": 0.10000002384185791, + "commonsense_qa": 0.20352172106504438, + "piqa": 0.3199129104614258, + "openbook_qa": 0.07200002670288086, + "lambada_openai": 0.31554433703422546, + "hellaswag": 0.13111595312754312, + "winograd": 0.09890115261077881, + "winogrande": 0.02131021022796631, + "bigbench_dyck_languages": 0.0650000050663948, + "agi_eval_lsat_ar": 0.10869564861059187, + "bigbench_cs_algorithms": 0.426515132188797, + "bigbench_operators": 0.10476190596818924, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.14692525565624237, + "coqa": 0.17825378477573395, + "boolq": -0.06872696625558952, + "bigbench_language_identification": 0.17744773842714012 + }, + "training_time_seconds": 6818.212191104889, + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/2c3bf17b", + "huggingface_url": "https://huggingface.co/jbduran/think-nanochat-d12/tree/main/experiments/climbmix-d12-1epoch-25shards", + "dataset_fingerprint": "29713846c3c08835", + "tokenizer_fingerprint": "285fd2719d7380ad", + "unique_train_tokens": 1321205760, + "effective_epochs": 1.0 +} diff --git a/experiments/climbmix-d12-1epoch-25shards/tokenizer/experiment_tokenizer.json b/experiments/climbmix-d12-1epoch-25shards/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..5c6b16fa95137e24b9355f413e2659d44189a44c --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/tokenizer/experiment_tokenizer.json @@ -0,0 +1,19 @@ +{ + "experiment_id": "climbmix-d12-1epoch-25shards", + "dataset": { + "adapter": "parquet_shards", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "base_url": "https://huggingface.co/datasets/karpathy/climbmix-400b-shuffle/resolve/main", + "validation_shard": 6542, + "num_train_shards": 25, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1781143229 +} diff --git a/experiments/climbmix-d12-1epoch-25shards/tokenizer/token_bytes.pt b/experiments/climbmix-d12-1epoch-25shards/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..9b3435c49ec7d41906a23e6147e57ff3fc94020d --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:009ef93d20dd4684497c19c4b7fed29278b53f20022d2dc39264e27b9eefa2a8 +size 132649 diff --git a/experiments/climbmix-d12-1epoch-25shards/tokenizer/tokenizer.pkl b/experiments/climbmix-d12-1epoch-25shards/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..019b5ce0de46ab6fb98b9b3ed0956aeed483fd44 --- /dev/null +++ b/experiments/climbmix-d12-1epoch-25shards/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ae73c5f7a960edc56022dfd46a653df2b9b38d84456e5e9f48eb5e02a60c21c2 +size 412126 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_000500.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..842f9e37a0db08c5ba32b6a13ac91bafb29a6a71 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_000500.json @@ -0,0 +1,186 @@ +{ + "step": 500, + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "val_bpb": 1.3171868025813407, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "ib80climb20-d12-1ep-26sh-r12", + "wandb_run_id": "cc999c41", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 12.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1, + "save_every": 500, + "model_tag": "ib80climb20-d12-1ep-26sh-r12", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" + }, + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.3171868025813407, + "smooth_train_loss": 3.7414761193419888, + "total_training_time": 1282.0482211112976, + "stage_training_flops": 232547388751872000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 232547388751872000 + } +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001000.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..345d6bc7b2870f71bd825acbcad144caa0ab7dcf --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001000.json @@ -0,0 +1,186 @@ +{ + "step": 1000, + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "val_bpb": 1.2342483403874234, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "ib80climb20-d12-1ep-26sh-r12", + "wandb_run_id": "cc999c41", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 12.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1, + "save_every": 500, + "model_tag": "ib80climb20-d12-1ep-26sh-r12", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" + }, + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 1.2342483403874234, + "smooth_train_loss": 3.5135694958986257, + "total_training_time": 2587.891883611679, + "stage_training_flops": 465094777503744000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 465094777503744000 + } +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001500.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..66ba21ca84abfe2818431eba6bb5e7722c9c061e --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_001500.json @@ -0,0 +1,186 @@ +{ + "step": 1500, + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "val_bpb": 1.1748804664299872, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "ib80climb20-d12-1ep-26sh-r12", + "wandb_run_id": "cc999c41", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 12.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1, + "save_every": 500, + "model_tag": "ib80climb20-d12-1ep-26sh-r12", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" + }, + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 1.1748804664299872, + "smooth_train_loss": 3.1845675293563995, + "total_training_time": 3895.9449546337128, + "stage_training_flops": 697642166255616000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 697642166255616000 + } +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002000.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..8c9e3b1867589272693573de8798cd5758a6152a --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002000.json @@ -0,0 +1,186 @@ +{ + "step": 2000, + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "val_bpb": 1.128405625513683, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "ib80climb20-d12-1ep-26sh-r12", + "wandb_run_id": "cc999c41", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 12.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1, + "save_every": 500, + "model_tag": "ib80climb20-d12-1ep-26sh-r12", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" + }, + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 1.128405625513683, + "smooth_train_loss": 3.155137224481085, + "total_training_time": 5204.380163908005, + "stage_training_flops": 930189555007488000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 930189555007488000 + } +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002500.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002500.json new file mode 100644 index 0000000000000000000000000000000000000000..ba830a7739906657bdc1001714fe348bd3275490 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002500.json @@ -0,0 +1,186 @@ +{ + "step": 2500, + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "val_bpb": 1.2390006511364335, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "ib80climb20-d12-1ep-26sh-r12", + "wandb_run_id": "cc999c41", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 12.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1, + "save_every": 500, + "model_tag": "ib80climb20-d12-1ep-26sh-r12", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" + }, + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 13, + "pos": 10792769, + "epoch": 1, + "pq_idx": 13, + "rg_idx": 10792769 + }, + "loop_state": { + "min_val_bpb": 1.128405625513683, + "smooth_train_loss": 3.283352044948622, + "total_training_time": 6508.467170000076, + "stage_training_flops": 1162736943759360000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1162736943759360000 + } +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002520.json b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002520.json new file mode 100644 index 0000000000000000000000000000000000000000..f4b44299b6be8a45ccdc67becfe28f7b38f13ee2 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/meta_002520.json @@ -0,0 +1,186 @@ +{ + "step": 2520, + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "val_bpb": 1.2371360397130737, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "ib80climb20-d12-1ep-26sh-r12", + "wandb_run_id": "cc999c41", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,climbmix,d12,1ep,26sh,r12,mix-80-20", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 12.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "experiment_config": "/content/nanochat_cache/experiments/ib80climb20-d12-1ep-26sh-r12/config.json", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1, + "save_every": 500, + "model_tag": "ib80climb20-d12-1ep-26sh-r12", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" + }, + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 13, + "pos": 21278849, + "epoch": 1, + "pq_idx": 13, + "rg_idx": 21278849 + }, + "loop_state": { + "min_val_bpb": 1.128405625513683, + "smooth_train_loss": 3.247208542986062, + "total_training_time": 6560.660916090012, + "stage_training_flops": 1172038839309434880, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1172038839309434880 + } +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_000500.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..4a6f5efe5ea389bcf4da07f147c3c15d51711df5 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0ee5c02fdfc77f8359298d79855fbdd1e3e25cfb4465aec43b504b4b330d0af1 +size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001000.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..624d0c04ece9410b6d1ffe52ced48a57dda933ff --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:4b6b89ed1d325c25bda441dd93e600faae27d96717a5756c3fe83add4edea20a +size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001500.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..c3533dbf65367af639a036dd9cd400f1a0544c6e --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1fa16ac0740c422f5867cfe6e2b312e1d9fe6f3c3093c4a562925d77973520dc +size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002000.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..e9a1da87bdfcab033f0b7700e0b0b614812b2ebb --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:80c5aefbead9aecd2f28352d73c18fac71d4f37ba129be6b2e02945206b1ea05 +size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002500.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002500.pt new file mode 100644 index 0000000000000000000000000000000000000000..a03a9376adccb70fa14e69f1f928f894d65c552e --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b5486ba7ac7781aef2d71fbf47f1e70d1d1cc2ce85583008d0f277c39c9c0d3 +size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002520.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002520.pt new file mode 100644 index 0000000000000000000000000000000000000000..74f19b5f0750c15eb7afdc0104a40d5157722e2c --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/model_002520.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a6f5dcb68f4be9513e13c428083e504c6df437d0e6ae3cbbf0aa2651587de86c +size 792761690 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_000500_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..953613094983098446ec00174b2e62741e01a626 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc23a7d1d8843b5b02703b96040f548eb8b3d72ed12a7187e5a93fcbeecbea5b +size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001000_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..6c2bfbc46eb4af8bdd430b416014f1944470dd7b --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:502fb4033c7a8e961223052edb7215378cadf96c0ed7a94f536f463ecc807ee6 +size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001500_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..407d540d2f40a8c484940b679755abece4c11803 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:26e19a82ec567925ba5e89d354c8d4b350bdcbe5de3d1b1be049a19810517941 +size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002000_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..493a635e49093fdcd5fe4b6f599912fb53064168 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8d0e35bf7141fc966f63ef557afd358b9a18f3c18842f51a214a42bf471f694b +size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002500_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b11e7bc4a101af6a343ce1f1a006f9f27b3d51e2 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5609a688d27407fae87a216e3d13a417375068420cdd5dbe5fc21a16434de284 +size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002520_rank0.pt b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002520_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..5f54ba46244e3a7eeeaf0cd48d159718ae1d64c0 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/base_checkpoints/optim_002520_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0e3a255abcbf36fb601466c2b4a87c8644a9f2c5f7fc8798097e090fc164012a +size 1246165357 diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/config.json b/experiments/ib80climb20-d12-1ep-26sh-r12/config.json new file mode 100644 index 0000000000000000000000000000000000000000..5b218744464f38e9c9696f4ed3efcb36634f61a0 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/config.json @@ -0,0 +1,105 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "datasets": [ + { + "name": "institutional-books", + "repo": "jbduran/think-dataset", + "revision": "main", + "ratio": 0.8, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7, + 8, + 9, + 10, + 11, + 12, + 13, + 14, + 15, + 16, + 17, + 18, + 19, + 20, + 21, + 22 + ], + "validation_shard": 472, + "download_workers": 4 + }, + { + "name": "climbmix", + "repo": "karpathy/climbmix-400b-shuffle", + "revision": "main", + "ratio": 0.2, + "train_shards": [ + 0, + 1, + 2, + 3, + 4, + 5, + 6, + 7 + ], + "validation_shard": 99, + "download_workers": 4 + } + ], + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 12.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "ib80climb20-d12-1ep-26sh-r12", + "group": "think-d12", + "tags": [ + "think-dataset", + "climbmix", + "d12", + "1ep", + "26sh", + "r12", + "mix-80-20" + ] + }, + "config_fingerprint": "fada94714cc2dbff", + "artifact_path": "experiments/ib80climb20-d12-1ep-26sh-r12" +} diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/core.json b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..d7627ff74411749b8248c61caa081b26bf04b3d8 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/core.json @@ -0,0 +1,56 @@ +{ + "model": "base_model (step 2520)", + "step": 2520, + "bpb": {}, + "core_metric": 0.10026160432378077, + "core_results": { + "hellaswag_zeroshot": 0.2837084233760834, + "jeopardy": 0.0009447330958209932, + "bigbench_qa_wikidata": 0.16987352073192596, + "arc_easy": 0.4431818127632141, + "arc_challenge": 0.24573378264904022, + "copa": 0.5, + "commonsense_qa": 0.2874692976474762, + "piqa": 0.6147986650466919, + "openbook_qa": 0.2760000228881836, + "lambada_openai": 0.29497379064559937, + "hellaswag": 0.28211510181427, + "winograd": 0.5970696210861206, + "winogrande": 0.5067087411880493, + "bigbench_dyck_languages": 0.026000000536441803, + "agi_eval_lsat_ar": 0.25652173161506653, + "bigbench_cs_algorithms": 0.415909081697464, + "bigbench_operators": 0.10000000149011612, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.13396404683589935, + "coqa": 0.10948264598846436, + "boolq": 0.5363914370536804, + "bigbench_language_identification": 0.2574999928474426 + }, + "centered_results": { + "hellaswag_zeroshot": 0.044944564501444496, + "jeopardy": 0.0009447330958209932, + "bigbench_qa_wikidata": 0.16987352073192596, + "arc_easy": 0.25757575035095215, + "arc_challenge": -0.005688289801279704, + "copa": 0.0, + "commonsense_qa": 0.10933662205934523, + "piqa": 0.2295973300933838, + "openbook_qa": 0.03466669718424479, + "lambada_openai": 0.29497379064559937, + "hellaswag": 0.042820135752360024, + "winograd": 0.1941392421722412, + "winogrande": 0.013417482376098633, + "bigbench_dyck_languages": 0.026000000536441803, + "agi_eval_lsat_ar": 0.07065216451883315, + "bigbench_cs_algorithms": 0.415909081697464, + "bigbench_operators": 0.10000000149011612, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.13396404683589935, + "coqa": 0.10948264598846436, + "boolq": -0.22002253406926203, + "bigbench_language_identification": 0.1831683089630832 + }, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/samples.json b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..4dd108d1b3a216f293c46031df4426da40cbdfcf --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 2520)", + "step": 2520, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is the capital of the world. The capital of the world is the capital of the" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is gold. It is a chemical element that is used to make gold. It is" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If you're not sure what to do, you can check out the" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is cold. The opposite of cold is cold. The opposite of cold is hot." + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: the sun, the moon, the sun, the moon, the sun, the" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the red. I like the red and the green. I like the red and" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" + } + ], + "unconditioned_samples": [ + "<|bos|>How to make a mock cryptic PR, a February 2016 article from the authors section of the \"Types of People\", can be found in the web site for more details.\n\nQuestion: What is one of the benefits of using ReinforcementAmit? Answer: ReinforcementAmit provides players with a healthy diet and strategies for successful business, since it exerts the widest selection of resources.", + "<|bos|>Chairman's Manual\n\nAdvertisement\n\nShareExplore\n\nASSISTY, MILLING DELAY\n\nEXECUTIVE INTERRANING THESE TI MINERALS DO NOT AGE\n\nJust compile a summary of the 18 endangered species of sea turtle from recently collected sea turtle remains in the estuarine and estuarine environments with the help of the National Oceanic and Atmospheric Administration (NOAA)", + "<|bos|>In my first time as a family I used to go to a lot of things to read and to get to know the basic components that allowed for healthy (good) hair and makeup for hair. I've never had it before, so it's fine to keep reading and if you top off your hair from time to time, that might be the reason.\n\nBut where I trace hair and makeup and make sure to get the right/Hair and hair profiles. And cutting the seams in the base of your chinboard is a good thing. And there are many bodices of the same breed as when you shine a little and pencil through your fingertips", + "<|bos|>Directions: 1) Fill a bucket with water in a large bowl. 2) Pour in the spices and seasonings. 3) In a bowl whisk together the diced tomatoes, onion, and garlic. 4) Place the diced bell peppers in a bowl and shake vigorously. Set aside.\n4) Introduce the diced tomatoes: Once the bell peppers have reached a firmness, they can be dropped rapidly. Place the peppers into the bowl and put a thought on them. Remember, they just never need a question.\n\nQuestion: What is the purpose of the rice balls and the peppers in the aforementioned mixture? Answer: To form", + "<|bos|>Dave's culinary debut was among the Falcons of the Snakes. It's still an easy pop-off into excess eating to the dogs, whether they're good or bad `` toggles ''. And one of the most noticeable aspects of Falcons is that they are long and tufted: one thing is for sure! Quick- And-Goopsy giving, step by step ; turning the buttons-lasts how group meal is structured.It's that easy, Step-by-step! The Falcons eat all the food together , and form the perfect organic whole - bite - part of their savory food.\nWhen Ozemne comes", + "<|bos|>Hardware\nE. Ten. Airds rayed and meshed in Annex 0 and 1\u2026\n\nExamples of PDF: Documents and Models, a comprehensive resource written and likewise clearly explained on the pieces page by an author in excellent colleagues.. Unlike part numbers, the original document reproduction is nethelix (to be more precise).", + "<|bos|>Well, you may want to know that this project also gives Oregon researchers an opportunity to learn how her stocks relate to the K-127. Three months later, you will find that the B-127 is providing access to a wealth of secrets about the K-127 that still have a long time to read.\n\nAs for how Oregon's secrets will affect her assets, you can look to the right tab below, just for a moment.\n\n1. Banejiq\n\nThis man was in Texas, while the other two brothers were UA's. What kept you from making the discovery you in the right place may be doubt", + "<|bos|>The sharing of information among grass producers gives farmers reason to believe they need to be able to handle climate change substantially better and, as part of their efforts to ensure the safety of agriculture while at the same time improving their health and fertility, it has set the tone for a research project in Prague, where it has revealed weather differences, so that the USDA beef farmer\u00b4s hours of practical knowledge have been strengthened.\n\nThough he lost farm productivity a few weeks ago, he had access to seasonal food staples like tomatoes, zucchini, cucumbers and cottage cheese and for the most part made friends with farmers from around the city. However, one year" + ] +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/evals/val_bpb.json b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..f2060a97326fd80f76ce2e6f89c780fc95096fa2 --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2520)", + "step": 2520, + "bpb": { + "val": 1.1701973227550237 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/run.json b/experiments/ib80climb20-d12-1ep-26sh-r12/run.json new file mode 100644 index 0000000000000000000000000000000000000000..cf97c574583a11bc7fd8e3493a7c274c4da4707c --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "fada94714cc2dbff", + "wandb_run_id": "cc999c41", + "created_at": 1781542858 +} diff --git a/experiments/ib80climb20-d12-1ep-26sh-r12/summary.json b/experiments/ib80climb20-d12-1ep-26sh-r12/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..ca8ff2ffa44c2b68954b87f616b9f23b32906b9b --- /dev/null +++ b/experiments/ib80climb20-d12-1ep-26sh-r12/summary.json @@ -0,0 +1,93 @@ +{ + "experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "stage": "base", + "base_experiment_id": "ib80climb20-d12-1ep-26sh-r12", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "jbduran/think-dataset+karpathy/climbmix-400b-shuffle", + "dataset_revision": null, + "step": 2520, + "depth": 12, + "target_param_data_ratio": 12.0, + "training_tokens": 1321205760, + "final_sampled_val_bpb": 1.2371360397130737, + "minimum_sampled_val_bpb": 1.128405625513683, + "full_val_bpb": 1.1701973227550237, + "core_metric": 0.10026160432378077, + "centered_results": { + "hellaswag_zeroshot": 0.044944564501444496, + "jeopardy": 0.0009447330958209932, + "bigbench_qa_wikidata": 0.16987352073192596, + "arc_easy": 0.25757575035095215, + "arc_challenge": -0.005688289801279704, + "copa": 0.0, + "commonsense_qa": 0.10933662205934523, + "piqa": 0.2295973300933838, + "openbook_qa": 0.03466669718424479, + "lambada_openai": 0.29497379064559937, + "hellaswag": 0.042820135752360024, + "winograd": 0.1941392421722412, + "winogrande": 0.013417482376098633, + "bigbench_dyck_languages": 0.026000000536441803, + "agi_eval_lsat_ar": 0.07065216451883315, + "bigbench_cs_algorithms": 0.415909081697464, + "bigbench_operators": 0.10000000149011612, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.13396404683589935, + "coqa": 0.10948264598846436, + "boolq": -0.22002253406926203, + "bigbench_language_identification": 0.1831683089630832 + }, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is the capital of the world. The capital of the world is the capital of the" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is gold. It is a chemical element that is used to make gold. It is" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday. If you're not sure what to do, you can check out the" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is cold. The opposite of cold is cold. The opposite of cold is hot." + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: the sun, the moon, the sun, the moon, the sun, the" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is the red. I like the red and the green. I like the red and" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is the number of times the number of times the number of times the number of times" + } + ], + "unconditioned_samples": [ + "<|bos|>How to make a mock cryptic PR, a February 2016 article from the authors section of the \"Types of People\", can be found in the web site for more details.\n\nQuestion: What is one of the benefits of using ReinforcementAmit? Answer: ReinforcementAmit provides players with a healthy diet and strategies for successful business, since it exerts the widest selection of resources.", + "<|bos|>Chairman's Manual\n\nAdvertisement\n\nShareExplore\n\nASSISTY, MILLING DELAY\n\nEXECUTIVE INTERRANING THESE TI MINERALS DO NOT AGE\n\nJust compile a summary of the 18 endangered species of sea turtle from recently collected sea turtle remains in the estuarine and estuarine environments with the help of the National Oceanic and Atmospheric Administration (NOAA)", + "<|bos|>In my first time as a family I used to go to a lot of things to read and to get to know the basic components that allowed for healthy (good) hair and makeup for hair. I've never had it before, so it's fine to keep reading and if you top off your hair from time to time, that might be the reason.\n\nBut where I trace hair and makeup and make sure to get the right/Hair and hair profiles. And cutting the seams in the base of your chinboard is a good thing. And there are many bodices of the same breed as when you shine a little and pencil through your fingertips", + "<|bos|>Directions: 1) Fill a bucket with water in a large bowl. 2) Pour in the spices and seasonings. 3) In a bowl whisk together the diced tomatoes, onion, and garlic. 4) Place the diced bell peppers in a bowl and shake vigorously. Set aside.\n4) Introduce the diced tomatoes: Once the bell peppers have reached a firmness, they can be dropped rapidly. Place the peppers into the bowl and put a thought on them. Remember, they just never need a question.\n\nQuestion: What is the purpose of the rice balls and the peppers in the aforementioned mixture? Answer: To form", + "<|bos|>Dave's culinary debut was among the Falcons of the Snakes. It's still an easy pop-off into excess eating to the dogs, whether they're good or bad `` toggles ''. And one of the most noticeable aspects of Falcons is that they are long and tufted: one thing is for sure! Quick- And-Goopsy giving, step by step ; turning the buttons-lasts how group meal is structured.It's that easy, Step-by-step! The Falcons eat all the food together , and form the perfect organic whole - bite - part of their savory food.\nWhen Ozemne comes", + "<|bos|>Hardware\nE. Ten. Airds rayed and meshed in Annex 0 and 1\u2026\n\nExamples of PDF: Documents and Models, a comprehensive resource written and likewise clearly explained on the pieces page by an author in excellent colleagues.. Unlike part numbers, the original document reproduction is nethelix (to be more precise).", + "<|bos|>Well, you may want to know that this project also gives Oregon researchers an opportunity to learn how her stocks relate to the K-127. Three months later, you will find that the B-127 is providing access to a wealth of secrets about the K-127 that still have a long time to read.\n\nAs for how Oregon's secrets will affect her assets, you can look to the right tab below, just for a moment.\n\n1. Banejiq\n\nThis man was in Texas, while the other two brothers were UA's. What kept you from making the discovery you in the right place may be doubt", + "<|bos|>The sharing of information among grass producers gives farmers reason to believe they need to be able to handle climate change substantially better and, as part of their efforts to ensure the safety of agriculture while at the same time improving their health and fertility, it has set the tone for a research project in Prague, where it has revealed weather differences, so that the USDA beef farmer\u00b4s hours of practical knowledge have been strengthened.\n\nThough he lost farm productivity a few weeks ago, he had access to seasonal food staples like tomatoes, zucchini, cucumbers and cottage cheese and for the most part made friends with farmers from around the city. However, one year" + ], + "training_time_seconds": 6560.660916090012, + "stage_training_flops": 1.172038839309435e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.172038839309435e+18, + "config_fingerprint": "fada94714cc2dbff", + "git_commit_sha": "85752bb3ef44571cd493d52ba59c7adbdd54d25a", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/cc999c41", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/ib80climb20-d12-1ep-26sh-r12", + "dataset_fingerprint": "643329665ca62fe0", + "tokenizer_fingerprint": "ff974929ec9d57e4", + "unique_train_tokens": 1360841933, + "effective_epochs": 0.9708737862650798 +} diff --git a/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/config.json b/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/config.json new file mode 100644 index 0000000000000000000000000000000000000000..12486bedcafce19f61c411e91cbe3757ae01875d --- /dev/null +++ b/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/config.json @@ -0,0 +1,50 @@ +{ + "schema_version": 1, + "stage": "sft", + "experiment_suffix": "complete-modern-sft-v1", + "parent": { + "base_experiment_id": "karpathy-nanochat-d34", + "checkpoint_step": 169150 + }, + "data": { + "recipe": "nanochat-default", + "mmlu_epochs": 3, + "gsm8k_epochs": 4, + "max_train_presentations": -1 + }, + "training": { + "num_iterations": -1, + "load_optimizer": 0, + "max_seq_len": 2048, + "device_batch_size": 4, + "total_batch_size": 524288, + "embedding_lr": 0.2, + "unembedding_lr": 0.004, + "matrix_lr": 0.02, + "init_lr_frac": 0.8, + "warmup_ratio": 0.0, + "warmdown_ratio": 0.5, + "final_lr_frac": 0.0, + "eval_every": -1, + "chatcore_every": -1, + "save_every": 200 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "enabled": true, + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "group": "karpathy-d34", + "tags": [ + "sft", + "karpathy-d34", + "nanochat-default", + "complete-mixture", + "fresh-optimizer" + ] + }, + "config_fingerprint": "d2c5a5ecd651045d", + "artifact_path": "experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1" +} diff --git a/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/run.json b/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/run.json new file mode 100644 index 0000000000000000000000000000000000000000..ae64c46c7675f84afd409f065cfc57c41367cc25 --- /dev/null +++ b/experiments/karpathy-nanochat-d34/sft/karpathy-nanochat-d34-complete-modern-sft-v1/run.json @@ -0,0 +1,11 @@ +{ + "experiment_id": "karpathy-nanochat-d34-complete-modern-sft-v1", + "stage": "sft", + "base_experiment_id": "karpathy-nanochat-d34", + "parent_experiment_id": "karpathy-nanochat-d34", + "parent_checkpoint_step": 169150, + "branch_parent_step": null, + "config_fingerprint": "d2c5a5ecd651045d", + "wandb_run_id": "b34a37b6", + "created_at": 1787102531 +} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_000500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..ae7bf02b4ea1eca6f729d198cc376edde5744feb --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_000500.json @@ -0,0 +1,140 @@ +{ + "step": 500, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.4440721103388112, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.4440721103388112, + "smooth_train_loss": 4.017187875737277, + "total_training_time": 1297.9931802749634, + "stage_training_flops": 232547388751872000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 232547388751872000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..1794dac81c418b665bf9a8357df29c84af7f6090 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001000.json @@ -0,0 +1,140 @@ +{ + "step": 1000, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.3600367137774247, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 1.3600367137774247, + "smooth_train_loss": 3.664376154963864, + "total_training_time": 2624.0160751342773, + "stage_training_flops": 465094777503744000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 465094777503744000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..c097ee41c745c78ae62449baf527ca275f99aed1 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_001500.json @@ -0,0 +1,140 @@ +{ + "step": 1500, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.3280401785061304, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 1.3280061850201952, + "smooth_train_loss": 3.6858135004402817, + "total_training_time": 3950.216495037079, + "stage_training_flops": 697642166255616000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 697642166255616000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..6e45dbda38d25705cb7dc4be3e80bdce10571852 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002000.json @@ -0,0 +1,140 @@ +{ + "step": 2000, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.2758348491157965, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 1.2758348491157965, + "smooth_train_loss": 3.5742909025197345, + "total_training_time": 5275.729542255402, + "stage_training_flops": 930189555007488000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 930189555007488000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002500.json new file mode 100644 index 0000000000000000000000000000000000000000..b59c4511fb50eb1d554602ea10ecce8a0d5cc2c4 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_002500.json @@ -0,0 +1,140 @@ +{ + "step": 2500, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.2427248605453876, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 13, + "pos": 10792769, + "epoch": 1, + "pq_idx": 13, + "rg_idx": 10792769 + }, + "loop_state": { + "min_val_bpb": 1.2427248605453876, + "smooth_train_loss": 3.3412281067532192, + "total_training_time": 6600.070840597153, + "stage_training_flops": 1162736943759360000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1162736943759360000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003000.json new file mode 100644 index 0000000000000000000000000000000000000000..f0b9b4e88828ab79c3b8e0cd68d73831f0e19a72 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003000.json @@ -0,0 +1,140 @@ +{ + "step": 3000, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.2025249805885556, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 15, + "pos": 72944769, + "epoch": 1, + "pq_idx": 15, + "rg_idx": 72944769 + }, + "loop_state": { + "min_val_bpb": 1.2025249805885556, + "smooth_train_loss": 3.3140674978000466, + "total_training_time": 7926.366828918457, + "stage_training_flops": 1395284332511232000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1395284332511232000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003500.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003500.json new file mode 100644 index 0000000000000000000000000000000000000000..74743aee2c51b33d3380795eb192137e1d77afdd --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_003500.json @@ -0,0 +1,140 @@ +{ + "step": 3500, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.1687947775345633, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 18, + "pos": 35096769, + "epoch": 1, + "pq_idx": 18, + "rg_idx": 35096769 + }, + "loop_state": { + "min_val_bpb": 1.1687947775345633, + "smooth_train_loss": 3.0847306468854994, + "total_training_time": 9253.729534626007, + "stage_training_flops": 1627831721263104000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1627831721263104000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004000.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004000.json new file mode 100644 index 0000000000000000000000000000000000000000..236ff6a99fb693d8710570316e441ede01a7209d --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004000.json @@ -0,0 +1,140 @@ +{ + "step": 4000, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.1246669836048595, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 20, + "pos": 97248769, + "epoch": 1, + "pq_idx": 20, + "rg_idx": 97248769 + }, + "loop_state": { + "min_val_bpb": 1.1246669836048595, + "smooth_train_loss": 3.0878954815260107, + "total_training_time": 10581.229209661484, + "stage_training_flops": 1860379110014976000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1860379110014976000 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004200.json b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004200.json new file mode 100644 index 0000000000000000000000000000000000000000..258b03f83162561d7473e8e2a3f668eb5e93e5be --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/meta_004200.json @@ -0,0 +1,140 @@ +{ + "step": 4200, + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "val_bpb": 1.1155106548699898, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "pre1900-d12-1ep-4sh-r20", + "wandb_run_id": "59d78984", + "wandb_group": "pre1900-d12", + "wandb_tags": "pre1900-corpus,d12,ratio20,4shards,1epoch", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 20.0, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "experiment_config": "/content/nanochat_cache/experiments/pre1900-d12-1ep-4sh-r20/config.json", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "pre1900-d12-1ep-4sh-r20", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" + }, + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 22, + "pos": 2109569, + "epoch": 1, + "pq_idx": 22, + "rg_idx": 2109569 + }, + "loop_state": { + "min_val_bpb": 1.1155106548699898, + "smooth_train_loss": 3.2450879718089913, + "total_training_time": 11112.257759571075, + "stage_training_flops": 1953398065515724800, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1953398065515724800 + } +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_000500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..af6edd914dc17156af26513c2577cd3d1049a7ba --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:348b87cda555d091e7fb42558bccaa2aa6dc1238dd6292b3826fffe639d3d5ac +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..bf76cd1fdaaeeaad85c74c46057fc00163b388c5 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:80b2dd0c88e2208837c7784d7453b1b5304cfe9eada1024626c91b945bcbe95b +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..08b0222d9ce3d9ee3590cc4d4e6dd7fd4a846082 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:30727a132972164e8f477d30e4258ee98bd747a6747ebc732deebb11eaf973a5 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..e72be4ec725a6d2a01f7ddb038e4a8ecc24b7fd9 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f6e98e913a7ce0b076a3a8e246bc0ab0369816071671bbe705287cfd09942f15 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002500.pt new file mode 100644 index 0000000000000000000000000000000000000000..684fe70424f2685f7efff277adc31090d06fd02b --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_002500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0ef0766f989c5016fcb1e3b1125acdf342adc931d6f24984e73b820fe939fb26 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003000.pt new file mode 100644 index 0000000000000000000000000000000000000000..190e5240a9235222fa8ac11e1088f07fd6926a7e --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f5227a1eff500535694508ed71c4d9802e7a85f2b562b006aaf289510559cf14 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003500.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003500.pt new file mode 100644 index 0000000000000000000000000000000000000000..d804ab27a38111cd4f4d3086285f005d97300a17 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_003500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:06e1588104aeda27382230a02f2c75869ea55a61889d014da02ebbe257988557 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004000.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004000.pt new file mode 100644 index 0000000000000000000000000000000000000000..d4dc33785e3f663b644442c576784d732569b54e --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c6763244f5f3349176b8af99aad687dc155f8ce81ee9c7a111ed6edf356487b6 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004200.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004200.pt new file mode 100644 index 0000000000000000000000000000000000000000..262525b39b3ae4cce7e86f17338503ed64300dfc --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/model_004200.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dd0bc3175b836b1306cac76c06f3660e26cf09bacc9f97d4b9dcba226712cda3 +size 792761690 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_000500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..697ca7be30689badd4ff00fdea8f29404dcdcd1b --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:de7d1085cac0ffcd273a5edb57c5c3c5eacbb6279b074c43dc687ad6c3c9a145 +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..e69bef4bb415e18568be9d00ed4d2d3756d78e0b --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b9e7a08c9a799a130d1a773eb8b74c987100048eb8155dabe930221aba039b83 +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..98b47d7a3702d17d9407d16b65e36f864a36f809 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92f4059f97359f2194043da27ed26c8d626af6afeed555fc92ab4a23a15460c4 +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b0ed4a9b664aff4de3639504ed8cb22245e392d3 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:33214aa57b70f8165ce7fe7fd6d75ec950e01bd58fcf802aa11af5a94ad6bf3a +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..add65ddce959c42044cd01849652913d360ff6ad --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_002500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8c0a51b3e3d4e809e224e02690c57ec626e3f02d18935a3dcebf5e14f184024a +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..4e963e5e3cf9ce2c144f788c2647301a326a6906 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e39341560c61a627558edaf0cdefbdaac71db46b193aafdb106ded04199a2d58 +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003500_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..a29c5235a7349aa44970919af689d9dbb6639aa8 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_003500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:be32cd5d76a6f2d5e4994d20a6d4e912209d9a3765879a1b6664d56931c38486 +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004000_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..61833e0fb3127c7b30569ef2d80c7889d27741ae --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e873b90b6248044760ba20db49d2310393019b7ab82946cd641274b9ffce449 +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004200_rank0.pt b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004200_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..f203f44e44258d8ff7a42e90f2c8160bb5f09cfd --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/base_checkpoints/optim_004200_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:23eaf1f3bdd7e46887ef59c43633560978293302b4d9326507a87a0b88c8766f +size 1246165357 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/config.json b/experiments/pre1900-d12-1ep-4sh-r20/config.json new file mode 100644 index 0000000000000000000000000000000000000000..e9b154ac56a36e262ba0803c4c41bc1104f004b4 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/config.json @@ -0,0 +1,58 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-4sh-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900-corpus", + "d12", + "ratio20", + "4shards", + "1epoch" + ] + }, + "config_fingerprint": "27dd1d0b7b2eff44", + "artifact_path": "experiments/pre1900-d12-1ep-4sh-r20" +} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/evals/core.json b/experiments/pre1900-d12-1ep-4sh-r20/evals/core.json new file mode 100644 index 0000000000000000000000000000000000000000..4210a671f0865e8339c49989400f858508f7ba14 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/evals/core.json @@ -0,0 +1,56 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": {}, + "core_metric": 0.08775698798334049, + "core_results": { + "hellaswag_zeroshot": 0.2801234722137451, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.10560503602027893, + "arc_easy": 0.3207070529460907, + "arc_challenge": 0.23037542402744293, + "copa": 0.550000011920929, + "commonsense_qa": 0.32186731696128845, + "piqa": 0.5446137189865112, + "openbook_qa": 0.2460000067949295, + "lambada_openai": 0.3044828176498413, + "hellaswag": 0.2790280878543854, + "winograd": 0.5457875728607178, + "winogrande": 0.5043409466743469, + "bigbench_dyck_languages": 0.12600000202655792, + "agi_eval_lsat_ar": 0.2869565188884735, + "bigbench_cs_algorithms": 0.38712120056152344, + "bigbench_operators": 0.08095238357782364, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.01825922355055809, + "coqa": 0.08142302185297012, + "boolq": 0.6039755344390869, + "bigbench_language_identification": 0.25130000710487366 + }, + "centered_results": { + "hellaswag_zeroshot": 0.04016462961832682, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.10560503602027893, + "arc_easy": 0.0942760705947876, + "arc_challenge": -0.026166101296742756, + "copa": 0.10000002384185791, + "commonsense_qa": 0.15233414620161054, + "piqa": 0.08922743797302246, + "openbook_qa": -0.005333324273427327, + "lambada_openai": 0.3044828176498413, + "hellaswag": 0.038704117139180504, + "winograd": 0.09157514572143555, + "winogrande": 0.008681893348693848, + "bigbench_dyck_languages": 0.12600000202655792, + "agi_eval_lsat_ar": 0.10869564861059187, + "bigbench_cs_algorithms": 0.38712120056152344, + "bigbench_operators": 0.08095238357782364, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.01825922355055809, + "coqa": 0.08142302185297012, + "boolq": -0.04216964621292916, + "bigbench_language_identification": 0.176347642579619 + }, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/evals/samples.json b/experiments/pre1900-d12-1ep-4sh-r20/evals/samples.json new file mode 100644 index 0000000000000000000000000000000000000000..e91a2b82b4a03f729d0daa919faa8392ad46b143 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/evals/samples.json @@ -0,0 +1,48 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": {}, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is now in the hands of the French, and the French in the hands of the" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of silver. The" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and then Saturday, and then Sunday, and then Monday, and then" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the hotter, and the hotter the hotter. The hotter is" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is a dark brown, with a light shade of green on the back, and a" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is 5, then x is 5, then x is 5, then" + } + ], + "unconditioned_samples": [ + "<|bos|>J. H. Moffatt, Manager.\n\nRAILLE FOR THE MILL.\n\nFor 1892 We're paying to the Post Office the last amount paid into the treasury each year for supplies extras on deposit at this Office:\n\nDtBuAHe FOR THE MILL\n\nfor The last year ,DEARICE post Office Address-which was Michigan State, WIS IN vising - Row of Cheap 75 per cent. cash.\n\nfactice for The last year.\n\nThe problem OF | the tariff Is 10 settle the difference between this country and 1.-THE United States, and this ques- ! tion Is open TO mental complication", + "<|bos|>has an amendment, including the spikes of the bills. The bill, which amends the eighth section, fixes \" ), those which, on motion in the two branches , of the Legislature, must continue to bud, but after the year 1781 shall again refer the ques- in the report of the Joint Select Com mittee on the Territories to select officers in each branch of the House annually provided until their successors are duly elected IN both branches.\n\n2ndIr i:\n\n}. That the Speaker of the House of Representatives, through the Sergeant at Arms, whose duty it is to nominate members OF Congress, shall in all cases select the", + "<|bos|>Again the clock on the chimneypiece strikes the half-hour in the intervals between the half hour and quarter struck by the clock on the house. House. He goes but will return again and tries the house, but the name and the two Cockneys will hate each other\n\nHeit among the neglected and fruitless in all the rambling and leafess sylvanism of a hne beach gossip among a crowd of dilapidated settlers-the eousel Mr. Shields might not turn house nor tree Every loop-hole is broken where battle signals disabled Every door is broken at every point where haste to resume the shuttle or effort to claP people. H", + "<|bos|>inteiful 100, it'll IL sour offi,a;naa pasualizing, form is subdivision of intellect as yo1aah customer and will be both increased and ornamented. The same rule applies 10 al work or source Of strength they are manipulated by the mind When svotper is WORKING IL W h questionable or impure WORK 10 not make specia believe, t94n intinct or working verb, but only 20 assert ability TO make believe 10 disbelieve, and this ability is AL nyISHtPO thoughts afish a,nhnrundian or rom Iast year.\n\nWHEN", + "<|bos|>DWsiuiuiu's LUeIy, with MIR. Grand Tor The defendant.\n\n| Total number OF witnesses by Mrs. | Waters, and she obtained their oath by Walkers | drinking, getting her divorce. | Willagln is only nine i'm bed. A wife is in Jail on charge of lntolerance.\n\nOn Monday the jury returned a verdict OF | murder against John Dora Blair. One OF McPherson's Chickadee Gus 'rais have begun enp!ling Greetings from Ella Benson, the proprietor OF the Pa. | ,ce Hotel.\n\n| To-day, TO the inquiry OF the Bank", + "<|bos|>GRADE\n\nE.W. UXBRIDGE ALPAHAEN\n\nWUXBRIDGE CARCS8) SILASHUEY STELLI AVE\n\nLike unto the fowler with touch of fire,\n\nWhen vivid the lightnings would flash desire:\n\nTheir awakshinc originareth the sky singeth ay the world, with embarrassment abounding aforefhem nameing opportunely to Belvideer Jews Revilejan hght.\n\nElbmergus Borringer.\n\nApppres sahoersativ schonwatter, Erickie frankel marafarsch | Walleton villn", + "<|bos|>rys substances of all percussion caps not ined, or mot has been lowered or made smate, or succeeded to the caloric of oSeH soot-cran, or from some other chief substance; otherwise it viii be temper-\n\nthe heat will not make Perseus little unless virtue and Wealth require it too, barely say newspaper paper! There be many lovers IN tilins But let them bear in mind the\n\nfirst principles in the selectione Of speeches, and in the popular orator, and rather in\n\nthe popular tale than the truth itself; so that where it\n\ndoes occur, as in familiar instances,", + "<|bos|>The figurative language would be appropriate in a town of town ships, but it is one of the affairs of the school to speak of the town home in the home, and is then one of the suggestions for those who are trying to find out how to understand the school term.\n\nThe figurative expressions would be jesuitous and those allusions purely aphoristic. It is only as much of the styles in play as hasn't the life really in it that it then bears. Unless the figurative language has a genuine article it is unattainable. where we have only the elastic spirit, but as much" + ] +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/evals/val_bpb.json b/experiments/pre1900-d12-1ep-4sh-r20/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..58b9124964272144bc9ff2e0bd2012405bfdbb00 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 4200)", + "step": 4200, + "bpb": { + "val": 1.080046782192013 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file diff --git a/experiments/pre1900-d12-1ep-4sh-r20/run.json b/experiments/pre1900-d12-1ep-4sh-r20/run.json new file mode 100644 index 0000000000000000000000000000000000000000..d4a8373fa87093fb4984f6a2bdb6ecd48251cef1 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "27dd1d0b7b2eff44", + "wandb_run_id": "59d78984", + "created_at": 1781914327 +} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/summary.json b/experiments/pre1900-d12-1ep-4sh-r20/summary.json new file mode 100644 index 0000000000000000000000000000000000000000..491cd7bafc30f4d2c821fa0d30d080d2bf8120d2 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/summary.json @@ -0,0 +1,93 @@ +{ + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-4sh-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "dataset": "mhla/pre1900-corpus", + "dataset_revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "step": 4200, + "depth": 12, + "target_param_data_ratio": 20.0, + "training_tokens": 2202009600, + "final_sampled_val_bpb": 1.1155106548699898, + "minimum_sampled_val_bpb": 1.1155106548699898, + "full_val_bpb": 1.080046782192013, + "core_metric": 0.08775698798334049, + "centered_results": { + "hellaswag_zeroshot": 0.04016462961832682, + "jeopardy": 0.0004723665479104966, + "bigbench_qa_wikidata": 0.10560503602027893, + "arc_easy": 0.0942760705947876, + "arc_challenge": -0.026166101296742756, + "copa": 0.10000002384185791, + "commonsense_qa": 0.15233414620161054, + "piqa": 0.08922743797302246, + "openbook_qa": -0.005333324273427327, + "lambada_openai": 0.3044828176498413, + "hellaswag": 0.038704117139180504, + "winograd": 0.09157514572143555, + "winogrande": 0.008681893348693848, + "bigbench_dyck_languages": 0.12600000202655792, + "agi_eval_lsat_ar": 0.10869564861059187, + "bigbench_cs_algorithms": 0.38712120056152344, + "bigbench_operators": 0.08095238357782364, + "bigbench_repeat_copy_logic": 0.0, + "squad": 0.01825922355055809, + "coqa": 0.08142302185297012, + "boolq": -0.04216964621292916, + "bigbench_language_identification": 0.176347642579619 + }, + "conditioned_samples": [ + { + "prompt": "The capital of France is", + "text": "<|bos|>The capital of France is now in the hands of the French, and the French in the hands of the" + }, + { + "prompt": "The chemical symbol of gold is", + "text": "<|bos|>The chemical symbol of gold is the same as that of silver, and the same as that of silver. The" + }, + { + "prompt": "If yesterday was Friday, then tomorrow will be", + "text": "<|bos|>If yesterday was Friday, then tomorrow will be Saturday, and then Saturday, and then Sunday, and then Monday, and then" + }, + { + "prompt": "The opposite of hot is", + "text": "<|bos|>The opposite of hot is the hotter, and the hotter the hotter. The hotter is" + }, + { + "prompt": "The planets of the solar system are:", + "text": "<|bos|>The planets of the solar system are: 1. The sun, 2. The moon, 3. The" + }, + { + "prompt": "My favorite color is", + "text": "<|bos|>My favorite color is a dark brown, with a light shade of green on the back, and a" + }, + { + "prompt": "If 5*x + 3 = 13, then x is", + "text": "<|bos|>If 5*x + 3 = 13, then x is 5, then x is 5, then x is 5, then" + } + ], + "unconditioned_samples": [ + "<|bos|>J. H. Moffatt, Manager.\n\nRAILLE FOR THE MILL.\n\nFor 1892 We're paying to the Post Office the last amount paid into the treasury each year for supplies extras on deposit at this Office:\n\nDtBuAHe FOR THE MILL\n\nfor The last year ,DEARICE post Office Address-which was Michigan State, WIS IN vising - Row of Cheap 75 per cent. cash.\n\nfactice for The last year.\n\nThe problem OF | the tariff Is 10 settle the difference between this country and 1.-THE United States, and this ques- ! tion Is open TO mental complication", + "<|bos|>has an amendment, including the spikes of the bills. The bill, which amends the eighth section, fixes \" ), those which, on motion in the two branches , of the Legislature, must continue to bud, but after the year 1781 shall again refer the ques- in the report of the Joint Select Com mittee on the Territories to select officers in each branch of the House annually provided until their successors are duly elected IN both branches.\n\n2ndIr i:\n\n}. That the Speaker of the House of Representatives, through the Sergeant at Arms, whose duty it is to nominate members OF Congress, shall in all cases select the", + "<|bos|>Again the clock on the chimneypiece strikes the half-hour in the intervals between the half hour and quarter struck by the clock on the house. House. He goes but will return again and tries the house, but the name and the two Cockneys will hate each other\n\nHeit among the neglected and fruitless in all the rambling and leafess sylvanism of a hne beach gossip among a crowd of dilapidated settlers-the eousel Mr. Shields might not turn house nor tree Every loop-hole is broken where battle signals disabled Every door is broken at every point where haste to resume the shuttle or effort to claP people. H", + "<|bos|>inteiful 100, it'll IL sour offi,a;naa pasualizing, form is subdivision of intellect as yo1aah customer and will be both increased and ornamented. The same rule applies 10 al work or source Of strength they are manipulated by the mind When svotper is WORKING IL W h questionable or impure WORK 10 not make specia believe, t94n intinct or working verb, but only 20 assert ability TO make believe 10 disbelieve, and this ability is AL nyISHtPO thoughts afish a,nhnrundian or rom Iast year.\n\nWHEN", + "<|bos|>DWsiuiuiu's LUeIy, with MIR. Grand Tor The defendant.\n\n| Total number OF witnesses by Mrs. | Waters, and she obtained their oath by Walkers | drinking, getting her divorce. | Willagln is only nine i'm bed. A wife is in Jail on charge of lntolerance.\n\nOn Monday the jury returned a verdict OF | murder against John Dora Blair. One OF McPherson's Chickadee Gus 'rais have begun enp!ling Greetings from Ella Benson, the proprietor OF the Pa. | ,ce Hotel.\n\n| To-day, TO the inquiry OF the Bank", + "<|bos|>GRADE\n\nE.W. UXBRIDGE ALPAHAEN\n\nWUXBRIDGE CARCS8) SILASHUEY STELLI AVE\n\nLike unto the fowler with touch of fire,\n\nWhen vivid the lightnings would flash desire:\n\nTheir awakshinc originareth the sky singeth ay the world, with embarrassment abounding aforefhem nameing opportunely to Belvideer Jews Revilejan hght.\n\nElbmergus Borringer.\n\nApppres sahoersativ schonwatter, Erickie frankel marafarsch | Walleton villn", + "<|bos|>rys substances of all percussion caps not ined, or mot has been lowered or made smate, or succeeded to the caloric of oSeH soot-cran, or from some other chief substance; otherwise it viii be temper-\n\nthe heat will not make Perseus little unless virtue and Wealth require it too, barely say newspaper paper! There be many lovers IN tilins But let them bear in mind the\n\nfirst principles in the selectione Of speeches, and in the popular orator, and rather in\n\nthe popular tale than the truth itself; so that where it\n\ndoes occur, as in familiar instances,", + "<|bos|>The figurative language would be appropriate in a town of town ships, but it is one of the affairs of the school to speak of the town home in the home, and is then one of the suggestions for those who are trying to find out how to understand the school term.\n\nThe figurative expressions would be jesuitous and those allusions purely aphoristic. It is only as much of the styles in play as hasn't the life really in it that it then bears. Unless the figurative language has a genuine article it is unattainable. where we have only the elastic spirit, but as much" + ], + "training_time_seconds": 11112.257759571075, + "stage_training_flops": 1.9533980655157248e+18, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1.9533980655157248e+18, + "config_fingerprint": "27dd1d0b7b2eff44", + "git_commit_sha": "48cd5f516123fcf119c7a43a32dd1dfd5c7b8eaa", + "wandb_url": "https://wandb.ai/jbduran-thinkingmachinesncsu/think.nano/runs/59d78984", + "huggingface_url": "https://huggingface.co/jbduran/think.nano/tree/main/experiments/pre1900-d12-1ep-4sh-r20", + "dataset_fingerprint": "b99d24e50808d3c5", + "tokenizer_fingerprint": "ab10cc8eda78637c", + "unique_train_tokens": 2268069888, + "effective_epochs": 0.970873786407767 +} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/experiment_tokenizer.json b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..5efd600c65d36b94b1fca1a6e5de698b995d6882 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "pre1900-d12-1ep-4sh-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 4, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1781914360 +} diff --git a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/token_bytes.pt b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..cc2638a46b3c5e02d8e316d6ceb532db60e5da83 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:98cdf539455a9dd9b63f5f4303da19912fb91d49dec505a80f01c4e13e60c38c +size 132649 diff --git a/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/tokenizer.pkl b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..567de529d282e5eda527aee448dbd68ca63f60f9 --- /dev/null +++ b/experiments/pre1900-d12-1ep-4sh-r20/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d4d66008f3953c4c06658f288439a51cc27967d15e0ffa730e154f4de3c3109e +size 398115 diff --git a/experiments/pre1900-d12-1ep-r20/config.json b/experiments/pre1900-d12-1ep-r20/config.json new file mode 100644 index 0000000000000000000000000000000000000000..03302b08d83c9ec554f0b8b3e63164c82c75185d --- /dev/null +++ b/experiments/pre1900-d12-1ep-r20/config.json @@ -0,0 +1,74 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "pre1900-d12-1ep-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 41, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "target_tokens": 2268069888, + "slack": 1.03, + "require_no_wrap": true, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "device_type": "cuda", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "scaling_params": 110100912, + "target_param_data_ratio": 20.0, + "window_pattern": "L", + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.28, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": 500, + "core_metric_max_per_task": 50, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "pre1900-d12-1ep-r20", + "group": "pre1900-d12", + "tags": [ + "pre1900", + "pre1900-corpus", + "d12", + "ratio20", + "r20", + "a100", + "bf16" + ] + }, + "config_fingerprint": "5faa20d56558c829", + "artifact_path": "experiments/pre1900-d12-1ep-r20" +} diff --git a/experiments/pre1900-d12-1ep-r20/run.json b/experiments/pre1900-d12-1ep-r20/run.json new file mode 100644 index 0000000000000000000000000000000000000000..bebf175cdd289fe98ebb9796ceff727d6956ef90 --- /dev/null +++ b/experiments/pre1900-d12-1ep-r20/run.json @@ -0,0 +1,10 @@ +{ + "experiment_id": "pre1900-d12-1ep-r20", + "stage": "base", + "base_experiment_id": "pre1900-d12-1ep-r20", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "5faa20d56558c829", + "wandb_run_id": "da8d0a59", + "created_at": 1781898389 +} diff --git a/experiments/pre1900-d12-1ep-r20/tokenizer/experiment_tokenizer.json b/experiments/pre1900-d12-1ep-r20/tokenizer/experiment_tokenizer.json new file mode 100644 index 0000000000000000000000000000000000000000..5284c308a121251d0c0a3b9f4a1bb23de95fec68 --- /dev/null +++ b/experiments/pre1900-d12-1ep-r20/tokenizer/experiment_tokenizer.json @@ -0,0 +1,18 @@ +{ + "experiment_id": "pre1900-d12-1ep-r20", + "dataset": { + "adapter": "parquet_shards", + "repo": "mhla/pre1900-corpus", + "revision": "ff332ef64f896c8d44672b4d1d0c752029e32f29", + "validation_shard": 41, + "num_train_shards": 41, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "created_at": 1781898707 +} diff --git a/experiments/pre1900-d12-1ep-r20/tokenizer/token_bytes.pt b/experiments/pre1900-d12-1ep-r20/tokenizer/token_bytes.pt new file mode 100644 index 0000000000000000000000000000000000000000..af63f54ce793582c533bb911725940394e3fec49 --- /dev/null +++ b/experiments/pre1900-d12-1ep-r20/tokenizer/token_bytes.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:93c2b481ec04debdcd92e3ad57b8ee5c0bfe806f6e85fc9aa1cb3111de4e6d84 +size 132649 diff --git a/experiments/pre1900-d12-1ep-r20/tokenizer/tokenizer.pkl b/experiments/pre1900-d12-1ep-r20/tokenizer/tokenizer.pkl new file mode 100644 index 0000000000000000000000000000000000000000..e78e2463ea2d42fe9eed62f1948f45ace1d6faf8 --- /dev/null +++ b/experiments/pre1900-d12-1ep-r20/tokenizer/tokenizer.pkl @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3c17e58522a9fd24956de85e56cd23d5f2131bb4a0d9af3433647fb8942add5b +size 397922 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_000500.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_000500.json new file mode 100644 index 0000000000000000000000000000000000000000..21925f13a4f63a1562871eba5b8138cad0cd7521 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_000500.json @@ -0,0 +1,142 @@ +{ + "step": 500, + "training_complete": false, + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "val_bpb": 1.3216810908332168, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-1ep-25sh-r11-wd42", + "wandb_run_id": "4e526b5a", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.42, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", + "tokenizer_fingerprint": "ebb3705d7792a34d", + "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-1ep-25sh-r11-wd42", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "weight_decay": 0.42, + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-1ep-25sh-r11-wd42", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11", + "25shards", + "1epoch", + "wd.42" + ] + }, + "config_fingerprint": "47bfa49108766b7d", + "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" + }, + "stage": "base", + "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "47bfa49108766b7d" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 2, + "pos": 62184769, + "epoch": 1, + "pq_idx": 2, + "rg_idx": 62184769 + }, + "loop_state": { + "min_val_bpb": 1.3216810908332168, + "smooth_train_loss": 3.7465076848716428, + "total_training_time": 1308.3161821365356, + "stage_training_flops": 232547388751872000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 232547388751872000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001000.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001000.json new file mode 100644 index 0000000000000000000000000000000000000000..1dc913105ff148b606108ba7d44ee6c225f5c541 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001000.json @@ -0,0 +1,142 @@ +{ + "step": 1000, + "training_complete": false, + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "val_bpb": 1.2426550595586172, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-1ep-25sh-r11-wd42", + "wandb_run_id": "4e526b5a", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.42, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", + "tokenizer_fingerprint": "ebb3705d7792a34d", + "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-1ep-25sh-r11-wd42", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "weight_decay": 0.42, + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-1ep-25sh-r11-wd42", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11", + "25shards", + "1epoch", + "wd.42" + ] + }, + "config_fingerprint": "47bfa49108766b7d", + "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" + }, + "stage": "base", + "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "47bfa49108766b7d" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 5, + "pos": 24336769, + "epoch": 1, + "pq_idx": 5, + "rg_idx": 24336769 + }, + "loop_state": { + "min_val_bpb": 1.2426550595586172, + "smooth_train_loss": 3.5992329357223416, + "total_training_time": 2646.5828285217285, + "stage_training_flops": 465094777503744000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 465094777503744000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001500.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001500.json new file mode 100644 index 0000000000000000000000000000000000000000..c41eed852092dd0dd83f743c0b2dfd677c12997f --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_001500.json @@ -0,0 +1,142 @@ +{ + "step": 1500, + "training_complete": false, + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "val_bpb": 1.1829778736744743, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-1ep-25sh-r11-wd42", + "wandb_run_id": "4e526b5a", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.42, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", + "tokenizer_fingerprint": "ebb3705d7792a34d", + "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-1ep-25sh-r11-wd42", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "weight_decay": 0.42, + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-1ep-25sh-r11-wd42", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11", + "25shards", + "1epoch", + "wd.42" + ] + }, + "config_fingerprint": "47bfa49108766b7d", + "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" + }, + "stage": "base", + "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "47bfa49108766b7d" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 7, + "pos": 86488769, + "epoch": 1, + "pq_idx": 7, + "rg_idx": 86488769 + }, + "loop_state": { + "min_val_bpb": 1.1829778736744743, + "smooth_train_loss": 3.4545608734819186, + "total_training_time": 3986.5272040367126, + "stage_training_flops": 697642166255616000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 697642166255616000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002000.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002000.json new file mode 100644 index 0000000000000000000000000000000000000000..0b829f3e6d4d0b956cf57e1b09a1873fd2b03447 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002000.json @@ -0,0 +1,142 @@ +{ + "step": 2000, + "training_complete": false, + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "val_bpb": 1.1274373944966156, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-1ep-25sh-r11-wd42", + "wandb_run_id": "4e526b5a", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.42, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", + "tokenizer_fingerprint": "ebb3705d7792a34d", + "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-1ep-25sh-r11-wd42", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "weight_decay": 0.42, + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-1ep-25sh-r11-wd42", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11", + "25shards", + "1epoch", + "wd.42" + ] + }, + "config_fingerprint": "47bfa49108766b7d", + "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" + }, + "stage": "base", + "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "47bfa49108766b7d" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 10, + "pos": 48640769, + "epoch": 1, + "pq_idx": 10, + "rg_idx": 48640769 + }, + "loop_state": { + "min_val_bpb": 1.1274373944966156, + "smooth_train_loss": 3.19770985990571, + "total_training_time": 5324.934033155441, + "stage_training_flops": 930189555007488000, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 930189555007488000 + } +} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002362.json b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002362.json new file mode 100644 index 0000000000000000000000000000000000000000..f63af5b1eb3294f4bbd7551c278ba6885908febe --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/meta_002362.json @@ -0,0 +1,142 @@ +{ + "step": 2362, + "training_complete": true, + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "val_bpb": 1.103670265641304, + "model_config": { + "sequence_len": 2048, + "vocab_size": 32768, + "n_layer": 12, + "n_head": 6, + "n_kv_head": 6, + "n_embd": 768, + "window_pattern": "L" + }, + "user_config": { + "run": "think-d12-1ep-25sh-r11-wd42", + "wandb_run_id": "4e526b5a", + "wandb_group": "think-d12", + "wandb_tags": "think-dataset,d12,ratio11,25shards,1epoch,wd.42", + "device_type": "", + "fp8": false, + "fp8_recipe": "tensorwise", + "depth": 12, + "aspect_ratio": 64, + "head_dim": 128, + "max_seq_len": 2048, + "window_pattern": "L", + "num_iterations": -1, + "target_flops": -1.0, + "target_param_data_ratio": 11.25, + "device_batch_size": 16, + "total_batch_size": 524288, + "embedding_lr": 0.3, + "unembedding_lr": 0.008, + "weight_decay": 0.42, + "matrix_lr": 0.02, + "scalar_lr": 0.5, + "warmup_steps": 40, + "warmdown_ratio": 0.65, + "final_lr_frac": 0.05, + "resume_from_step": -1, + "pretokenized": true, + "data_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/data", + "tokenizer_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/tokenizer", + "pretokenized_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/pretok", + "checkpoint_dir": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "experiment_config": "/content/nanochat_cache/experiments/think-d12-1ep-25sh-r11-wd42/config.json", + "tokenizer_fingerprint": "ebb3705d7792a34d", + "git_commit_sha": "7e29503cca7b67e1418323c3628e6f935e4feacf", + "seed": 42, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "core_metric_max_per_task": 500, + "sample_every": -1, + "save_every": 500, + "model_tag": "think-d12-1ep-25sh-r11-wd42", + "experiment": { + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "weight_decay": 0.42, + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-1ep-25sh-r11-wd42", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11", + "25shards", + "1epoch", + "wd.42" + ] + }, + "config_fingerprint": "47bfa49108766b7d", + "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" + }, + "stage": "base", + "base_experiment_id": "think-d12-1ep-25sh-r11-wd42", + "parent_experiment_id": null, + "parent_checkpoint_step": null, + "config_fingerprint": "47bfa49108766b7d" + }, + "device_batch_size": 16, + "max_seq_len": 2048, + "total_batch_size": 524288, + "dataloader_state_dict": { + "file_idx": 12, + "pos": 38438817, + "epoch": 1, + "pq_idx": 12, + "rg_idx": 38438817 + }, + "loop_state": { + "min_val_bpb": 1.103670265641304, + "smooth_train_loss": 3.1383014196569707, + "total_training_time": 6293.967695713043, + "stage_training_flops": 1098553864463843328, + "inherited_parent_flops": 0.0, + "cumulative_pipeline_training_flops": 1098553864463843328 + } +} \ No newline at end of file diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_000500.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_000500.pt new file mode 100644 index 0000000000000000000000000000000000000000..23f325a09c26660d57660a524dd13516c47bb557 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_000500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2ba7f29f5abc6b63320f4acbaf91d9bfdb0a6cfb5ec5214e22fd4b4e7cacde56 +size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001000.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001000.pt new file mode 100644 index 0000000000000000000000000000000000000000..e51cf0f090eec510c66d82c0eee4a292bbf950df --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2cf5c5d3bab8b0a98284389f71e40abef1a253000e05ba9c9bac41931767af40 +size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001500.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001500.pt new file mode 100644 index 0000000000000000000000000000000000000000..c4b02f48e6ddb7ac6fdb778a5fc2408097150054 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_001500.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47130e6261cf39ddc15a667483465368a1ede8fd96d4d4b1c75c4bf3deb1a8cc +size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002000.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002000.pt new file mode 100644 index 0000000000000000000000000000000000000000..50164f1c86395a7295ec8feb0306ffb8dda49456 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002000.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e1c858d0142791d42fcab3b5fbbd72d188032bc399f33da736ddf1ef864d5c0e +size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002362.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002362.pt new file mode 100644 index 0000000000000000000000000000000000000000..df8a17957e7c45b557bd477ccd73d02694d6b737 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/model_002362.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8b2299fe338c35e226e044e2f3555c7ec5cc6fd8d59317db2235590d01aef9d1 +size 792761690 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_000500_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_000500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..f1f1ca6963076c7bf1d346af21583cd2e0e273a7 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_000500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f37e7f554852ced4488a09521f28a55c40c1708feb4a969c82444cc52866cb9d +size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001000_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..b53953acb1eb6a9440d07528f9f0860efc8f86e9 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:45b9c62b9529b4e3da87ef073be54ea34562194f9810af830675fadb2d2dc2d1 +size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001500_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001500_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..52ae32f37959deed46455181bdd09f00ec1cb3fb --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_001500_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9d18f0823e9354165fbbe28dff83bd45cd3cface494ae994a5465340f8c23109 +size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002000_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002000_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..59e834a6b7188d20c84621317487f11a001a71ad --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002000_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:13abe8ff459d4a01082c70ed29497a8af2747cc3e94c7c298c3e83ff568048b2 +size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002362_rank0.pt b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002362_rank0.pt new file mode 100644 index 0000000000000000000000000000000000000000..17bdc04253494be5723658c443c83c965a7002de --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/base_checkpoints/optim_002362_rank0.pt @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8607ee156d616769b388c071de7dd3bc068f04fb8196f81843357ddb59c71dc2 +size 1246165357 diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/config.json b/experiments/think-d12-1ep-25sh-r11-wd42/config.json new file mode 100644 index 0000000000000000000000000000000000000000..6dcbeacb7abdcf69a3092a24acfe5f8092defa70 --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/config.json @@ -0,0 +1,59 @@ +{ + "schema_version": 1, + "stage": "base", + "experiment_id": "think-d12-1ep-25sh-r11-wd42", + "dataset": { + "adapter": "parquet_shards", + "repo": "jbduran/think-dataset", + "revision": "main", + "validation_shard": 472, + "num_train_shards": 44, + "download_workers": 4 + }, + "tokenizer": { + "mode": "train", + "max_chars": 2000000000, + "doc_cap": 10000, + "vocab_size": 32768 + }, + "pretokenize": { + "enabled": true, + "slack": 1.03, + "val_tokens": 20971520, + "shard_tokens": 100000000, + "tokenizer_threads": 8 + }, + "training": { + "depth": 12, + "scaling_params": 110100912, + "target_param_data_ratio": 11.25, + "window_pattern": "L", + "weight_decay": 0.42, + "device_batch_size": 16, + "total_batch_size": 524288, + "save_every": 500, + "eval_every": 250, + "eval_tokens": 2097152, + "core_metric_every": -1, + "sample_every": -1 + }, + "artifacts": { + "repo": "jbduran/think.nano" + }, + "wandb": { + "entity": "jbduran-thinkingmachinesncsu", + "project": "think.nano", + "name": "think-d12-1ep-25sh-r11-wd42", + "group": "think-d12", + "tags": [ + "think-dataset", + "d12", + "ratio11", + "25shards", + "1epoch", + "wd.42" + ] + }, + "config_fingerprint": "47bfa49108766b7d", + "artifact_path": "experiments/think-d12-1ep-25sh-r11-wd42" +} diff --git a/experiments/think-d12-1ep-25sh-r11-wd42/evals/val_bpb.json b/experiments/think-d12-1ep-25sh-r11-wd42/evals/val_bpb.json new file mode 100644 index 0000000000000000000000000000000000000000..9611aa080f82197f0797867dbca4c42b99fb733d --- /dev/null +++ b/experiments/think-d12-1ep-25sh-r11-wd42/evals/val_bpb.json @@ -0,0 +1,12 @@ +{ + "model": "base_model (step 2362)", + "step": 2362, + "bpb": { + "val": 1.0526348691238439 + }, + "core_metric": null, + "core_results": null, + "centered_results": null, + "conditioned_samples": [], + "unconditioned_samples": [] +} \ No newline at end of file