diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..7d870d5d5532ee64327c74b3ae4625aca1fbc678 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0d45b60c3377d2b233fb79150e4c6377564419264b3e0dd37b143f4dea3eeaca +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..ae9d1dd9bd9f40d28570f213271e02c8007884ef --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d914723f4cff80df3e933a3d63ed043f08536b104877dd055d63014f403de212 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..067666c96d57647af15b9172a0b8c14d0a18f5cf --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:af842a7f9e95aa8f8e6a126bead7c15243b1d3dd91e7dd5062278c436136c8d6 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..23b3d46da14fee4b9971453aa641c15f647582a4 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b7608843e5153961ba8c3edc1f197c34164f0867a809f0207e1008fecf20a06a +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..fb7a1c55b2239d507664c916b5f5c3d1081436d4 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:44243ca3716547c0c21a27577345ceaeb9e623be86ced11e1044d7ab6fa8a7e7 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..54fbdcfef68daf299c1a7a4c95ca9d3d2f2aef36 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ea2112c3636246412e69c2fcd14b814cd7b68334b9c3f8bbddb5b8208429e04a +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..6ba4e4d50e853148e9d60ef65243953ed03f963c --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1b8c7efe671d91efab26d3fbdc245783660112915cbb1d22a44ed154e94e2748 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..beba9ed658e3e1a25a91f08ff25db64d1906e81a --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:291f338c17fa2a84cb07d5db8619e1333d6e2f8944c902a3261c101b354ba731 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..97d2391e209c550ce4b08aadac16ff6a66d3ef96 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 10.044001579284668, + "learning_rate": 2e-05, + "loss": 1.0811, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 5.708449363708496, + "learning_rate": 2e-05, + "loss": 1.0935, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 6.164332389831543, + "learning_rate": 2e-05, + "loss": 0.2887, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 0.7329395413398743, + "learning_rate": 2e-05, + "loss": 0.0654, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 3.7981791496276855, + "learning_rate": 2e-05, + "loss": 0.5255, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 10.993570327758789, + "learning_rate": 2e-05, + "loss": 0.7257, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 6.135924816131592, + "learning_rate": 2e-05, + "loss": 0.3716, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 10.345540046691895, + "learning_rate": 2e-05, + "loss": 1.132, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 15.495490074157715, + "learning_rate": 2e-05, + "loss": 1.7626, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 9.968962669372559, + "learning_rate": 2e-05, + "loss": 1.0484, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 0.6021881103515625, + "learning_rate": 2e-05, + "loss": 0.0195, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 10.388916015625, + "learning_rate": 2e-05, + "loss": 0.813, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 6.319398880004883, + "learning_rate": 2e-05, + "loss": 0.2256, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 0.6123639345169067, + "learning_rate": 2e-05, + "loss": 0.1623, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 2.0997838973999023, + "learning_rate": 2e-05, + "loss": 0.8552, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 17.500144958496094, + "learning_rate": 2e-05, + "loss": 1.1279, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 5.181461334228516, + "learning_rate": 2e-05, + "loss": 0.2672, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 10.427047729492188, + "learning_rate": 2e-05, + "loss": 0.5116, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 4.001834392547607, + "learning_rate": 2e-05, + "loss": 0.8254, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 4.918587684631348, + "learning_rate": 2e-05, + "loss": 0.2577, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 1.5070538520812988, + "learning_rate": 2e-05, + "loss": 0.0786, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 8.23068904876709, + "learning_rate": 2e-05, + "loss": 1.336, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 0.4960165321826935, + "learning_rate": 2e-05, + "loss": 0.2786, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 14.814494132995605, + "learning_rate": 2e-05, + "loss": 1.3887, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 2.1397197246551514, + "learning_rate": 2e-05, + "loss": 0.1331, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 11.526412963867188, + "learning_rate": 2e-05, + "loss": 0.4426, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 2.414958953857422, + "learning_rate": 2e-05, + "loss": 0.2413, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 4.718291282653809, + "learning_rate": 2e-05, + "loss": 0.1895, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 12.340472221374512, + "learning_rate": 2e-05, + "loss": 1.9869, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 4.813292503356934, + "learning_rate": 2e-05, + "loss": 0.1454, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 2.3212244510650635, + "learning_rate": 2e-05, + "loss": 0.4304, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 3.7471375465393066, + "learning_rate": 2e-05, + "loss": 0.1793, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 0.6779154539108276, + "learning_rate": 2e-05, + "loss": 0.1912, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 5.2416229248046875, + "learning_rate": 2e-05, + "loss": 0.7198, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 12.138299942016602, + "learning_rate": 2e-05, + "loss": 0.6979, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 8.756268501281738, + "learning_rate": 2e-05, + "loss": 0.3211, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 12.594830513000488, + "learning_rate": 2e-05, + "loss": 2.7682, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 5.925561904907227, + "learning_rate": 2e-05, + "loss": 0.7106, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 0.4183579981327057, + "learning_rate": 2e-05, + "loss": 0.9514, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 6.107239246368408, + "learning_rate": 2e-05, + "loss": 1.1018, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 6.866254806518555, + "learning_rate": 2e-05, + "loss": 0.6412, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 13.476661682128906, + "learning_rate": 2e-05, + "loss": 0.9713, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 0.8483925461769104, + "learning_rate": 2e-05, + "loss": 0.1767, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 9.308485984802246, + "learning_rate": 2e-05, + "loss": 1.2229, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 4.227640151977539, + "learning_rate": 2e-05, + "loss": 0.4012, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 0.6228331923484802, + "learning_rate": 2e-05, + "loss": 0.9619, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 9.9182767868042, + "learning_rate": 2e-05, + "loss": 2.1173, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 3.4465816020965576, + "learning_rate": 2e-05, + "loss": 1.605, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 12.18694019317627, + "learning_rate": 2e-05, + "loss": 2.0198, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 1.0387413501739502, + "learning_rate": 2e-05, + "loss": 0.2066, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 2053257199353856.0, + "train_loss": 0.755521228313446, + "train_runtime": 101.7028, + "train_samples_per_second": 3.933, + "train_steps_per_second": 0.983 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2053257199353856.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..a3e7f1bdc039c9103df345744bb709a2c9edb513 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:bf0af71fd4d6dc12cffd01523df4bbc58a571c0978ca4c7d7571aa24bdeb8ee5 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..86ef678a775e7ca28a99403a265f1afc13cacb1a --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1744e2ae1501079d9883e05d012fe5bf1d7a4f72884671f31cc3a124d304ec02 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..715a90d3af60602b1aa6683fa2d537e31759956f --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c105cc5e4f96671ece69b8f9d708aa9a805d9a65a48b0865cc54c7d7fc57fdbf +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..db1f81c5bdea0994ca69f0beb95a6d6afc777060 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ff0a15058359453e6d079b47f6e618fcf9720db11d691877dbe2140c6a15be3c +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..a4f2b2deb6a64520937f91b370638b8fe39c5e9d --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1ec3b7d6ad6a720fb656550c2f62665f56860a6f8a8288ff1f826b9dceaed89e +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..dd75619c4e77c7d16e4fe5540d316a29dc209386 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:59993217d193880b0fc5548c04f04df882e9949136543c7bbede6e4dcfb2687c +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..6c4811c41c95c876a6bab828a837f05f15d603a0 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:009defbeb3403fee4f1ed358ade2be16ece2a7c97a4f95464240ea4cf6c2e3d8 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..c2dc6c8bf124f8d967dd0e1c48bed05e6470295d --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c0be9e6109694003abb07f24a9bb7b03b63669d481039c2e19e40b6296e98501 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..3f538efe59f01f293b553ba218b2212bf0d09ac5 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 0.02960115671157837, + "learning_rate": 2e-05, + "loss": 0.0045, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 0.09186350554227829, + "learning_rate": 2e-05, + "loss": 0.0653, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 0.16298067569732666, + "learning_rate": 2e-05, + "loss": 0.0063, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 0.09170705080032349, + "learning_rate": 2e-05, + "loss": 0.0034, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 0.06864725798368454, + "learning_rate": 2e-05, + "loss": 0.0031, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 0.19071757793426514, + "learning_rate": 2e-05, + "loss": 0.0287, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 0.17343077063560486, + "learning_rate": 2e-05, + "loss": 0.0038, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 0.2856125235557556, + "learning_rate": 2e-05, + "loss": 0.0591, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 0.04609253257513046, + "learning_rate": 2e-05, + "loss": 0.024, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 0.01641538180410862, + "learning_rate": 2e-05, + "loss": 0.001, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 0.07860966771841049, + "learning_rate": 2e-05, + "loss": 0.0113, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 0.027962295338511467, + "learning_rate": 2e-05, + "loss": 0.0011, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 1.3819764852523804, + "learning_rate": 2e-05, + "loss": 0.0356, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 0.01595745049417019, + "learning_rate": 2e-05, + "loss": 0.0008, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 0.12123987823724747, + "learning_rate": 2e-05, + "loss": 0.004, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 0.7385960221290588, + "learning_rate": 2e-05, + "loss": 0.1151, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 5.604942798614502, + "learning_rate": 2e-05, + "loss": 0.1588, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 0.005619928240776062, + "learning_rate": 2e-05, + "loss": 0.0003, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 0.027213625609874725, + "learning_rate": 2e-05, + "loss": 0.0016, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 4.135501384735107, + "learning_rate": 2e-05, + "loss": 0.0975, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 0.00542866624891758, + "learning_rate": 2e-05, + "loss": 0.0006, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 7.716265678405762, + "learning_rate": 2e-05, + "loss": 0.3077, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 0.01432905811816454, + "learning_rate": 2e-05, + "loss": 0.0009, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 0.3185250163078308, + "learning_rate": 2e-05, + "loss": 0.004, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 0.00824733730405569, + "learning_rate": 2e-05, + "loss": 0.0022, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 0.006523216143250465, + "learning_rate": 2e-05, + "loss": 0.0265, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 0.3274787664413452, + "learning_rate": 2e-05, + "loss": 0.0094, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 0.012340985238552094, + "learning_rate": 2e-05, + "loss": 0.0011, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 0.01708126626908779, + "learning_rate": 2e-05, + "loss": 0.6785, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 0.005370909348130226, + "learning_rate": 2e-05, + "loss": 0.2075, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 0.009917240589857101, + "learning_rate": 2e-05, + "loss": 0.0011, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 0.015617340803146362, + "learning_rate": 2e-05, + "loss": 0.0178, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 0.04841961711645126, + "learning_rate": 2e-05, + "loss": 0.0013, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 2.055001735687256, + "learning_rate": 2e-05, + "loss": 0.0487, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 0.010724334977567196, + "learning_rate": 2e-05, + "loss": 0.0007, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 0.1112007200717926, + "learning_rate": 2e-05, + "loss": 0.0032, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 0.5635241270065308, + "learning_rate": 2e-05, + "loss": 0.0103, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 0.024422986432909966, + "learning_rate": 2e-05, + "loss": 0.0009, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 9.877224922180176, + "learning_rate": 2e-05, + "loss": 0.4788, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 0.025433897972106934, + "learning_rate": 2e-05, + "loss": 0.0017, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 0.07279513031244278, + "learning_rate": 2e-05, + "loss": 0.0021, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 0.003639621427282691, + "learning_rate": 2e-05, + "loss": 0.0004, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 0.021531013771891594, + "learning_rate": 2e-05, + "loss": 1.1024, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 0.041703518480062485, + "learning_rate": 2e-05, + "loss": 0.0053, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 1.0306504964828491, + "learning_rate": 2e-05, + "loss": 0.0258, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 0.19764401018619537, + "learning_rate": 2e-05, + "loss": 0.0066, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 0.1398591846227646, + "learning_rate": 2e-05, + "loss": 0.0036, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 0.9593293070793152, + "learning_rate": 2e-05, + "loss": 0.0234, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 0.013573438860476017, + "learning_rate": 2e-05, + "loss": 0.0012, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 0.1273522675037384, + "learning_rate": 2e-05, + "loss": 0.003, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 2069634366832640.0, + "train_loss": 0.07204394578933716, + "train_runtime": 101.8827, + "train_samples_per_second": 3.926, + "train_steps_per_second": 0.982 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2069634366832640.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..5e360bb3b61d0353e8cd8bfed413c472493b1357 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2cd1af5c7ca40ed6a79af8c9cb6555f78589b0a624a98c6483565f76f8b3c61d +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..51c281c1fc7c1826002ca2df1569c5dde67688a6 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:8e8db792016ac55ff57bd23f8349e5df6ceb14624e1d9a29105c81ed7d7ed4ed +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..6fcbec7952fb9d959be011fbffce119e05716c98 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:19524fc38015f33290f292a4f522b4d36c1adcb621ac578bb561050e13fefe91 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..160ec64cd29c0e78f47a5a45e4a4d8705b97ceec --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:3a874d7d599aad5817e884d0953977542823faf1f01d25101400d3430928ff9c +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..f7c98fca20e49322bd9e8ccc52ce37ade7171177 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:33baa7cb694bc7b9f9162330827005d81631575ea2e90cd715b6d42b50f475da +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..e7ccf8f9310363dc996fc1d520afc9a74b734929 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cb3be0c563957173b91a7ec86c8723fdb23acdadf722cf602bc3483c5d0b11ae +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..3736050fe3f44af1498b2a433eca06bc26df8cfa --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:698b9afe0148feecdd937cdc3624b381dfe48c2a3a8083e2dfa29bfedd728c42 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..2136b8e130c9f55a31b07896326a44f6dbe91f14 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:633deab3c32fa4046bc1e4c0691c368ae244561291e78cbdba6062fdf7b77f16 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..7b89af726a08b198c89f1ad4113f29c05b66c7ac --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 1.765663981437683, + "learning_rate": 2e-05, + "loss": 0.1386, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 4.676456928253174, + "learning_rate": 2e-05, + "loss": 1.1946, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 2.184738874435425, + "learning_rate": 2e-05, + "loss": 0.9233, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 4.627042770385742, + "learning_rate": 2e-05, + "loss": 1.1015, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 5.915734767913818, + "learning_rate": 2e-05, + "loss": 0.663, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 6.462231636047363, + "learning_rate": 2e-05, + "loss": 1.0081, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 0.35552021861076355, + "learning_rate": 2e-05, + "loss": 0.3945, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 5.338579177856445, + "learning_rate": 2e-05, + "loss": 0.6744, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 1.783922791481018, + "learning_rate": 2e-05, + "loss": 0.4461, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 3.711841106414795, + "learning_rate": 2e-05, + "loss": 1.4118, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 0.9069531559944153, + "learning_rate": 2e-05, + "loss": 0.176, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 7.7085490226745605, + "learning_rate": 2e-05, + "loss": 0.5315, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 10.80366039276123, + "learning_rate": 2e-05, + "loss": 1.0346, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 6.07086181640625, + "learning_rate": 2e-05, + "loss": 0.3799, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 1.8229091167449951, + "learning_rate": 2e-05, + "loss": 0.1932, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 5.323653697967529, + "learning_rate": 2e-05, + "loss": 0.3793, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 4.461909770965576, + "learning_rate": 2e-05, + "loss": 0.6431, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 1.6867424249649048, + "learning_rate": 2e-05, + "loss": 0.1087, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 3.3799192905426025, + "learning_rate": 2e-05, + "loss": 0.6874, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 7.544458389282227, + "learning_rate": 2e-05, + "loss": 0.6817, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 2.379153251647949, + "learning_rate": 2e-05, + "loss": 0.6136, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 2.67325758934021, + "learning_rate": 2e-05, + "loss": 0.1987, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 1.6791194677352905, + "learning_rate": 2e-05, + "loss": 0.3694, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 0.7336940169334412, + "learning_rate": 2e-05, + "loss": 0.3386, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 2.389741897583008, + "learning_rate": 2e-05, + "loss": 0.1778, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 7.384143829345703, + "learning_rate": 2e-05, + "loss": 0.7985, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 7.0400872230529785, + "learning_rate": 2e-05, + "loss": 0.6097, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 8.808256149291992, + "learning_rate": 2e-05, + "loss": 0.7863, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 3.6529507637023926, + "learning_rate": 2e-05, + "loss": 0.1368, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 0.17764848470687866, + "learning_rate": 2e-05, + "loss": 0.1653, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 1.8120052814483643, + "learning_rate": 2e-05, + "loss": 0.7154, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 3.756453037261963, + "learning_rate": 2e-05, + "loss": 0.5272, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 3.120513677597046, + "learning_rate": 2e-05, + "loss": 0.1667, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 4.981636047363281, + "learning_rate": 2e-05, + "loss": 0.3927, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 5.101208209991455, + "learning_rate": 2e-05, + "loss": 0.6766, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 1.0963003635406494, + "learning_rate": 2e-05, + "loss": 0.7616, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 0.7717503905296326, + "learning_rate": 2e-05, + "loss": 0.0356, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 0.8139798641204834, + "learning_rate": 2e-05, + "loss": 0.1705, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 3.033977746963501, + "learning_rate": 2e-05, + "loss": 0.1349, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 13.496634483337402, + "learning_rate": 2e-05, + "loss": 1.6624, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 13.66792106628418, + "learning_rate": 2e-05, + "loss": 2.9498, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 0.06267403066158295, + "learning_rate": 2e-05, + "loss": 0.0111, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 5.407293319702148, + "learning_rate": 2e-05, + "loss": 0.5322, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 6.728641510009766, + "learning_rate": 2e-05, + "loss": 0.6077, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 5.393722057342529, + "learning_rate": 2e-05, + "loss": 1.6041, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 0.15307481586933136, + "learning_rate": 2e-05, + "loss": 0.3262, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 6.039478302001953, + "learning_rate": 2e-05, + "loss": 0.491, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 3.6059703826904297, + "learning_rate": 2e-05, + "loss": 0.2965, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 10.516265869140625, + "learning_rate": 2e-05, + "loss": 1.887, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 2.030522584915161, + "learning_rate": 2e-05, + "loss": 0.1721, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 4914533793529856.0, + "train_loss": 0.6217503023147583, + "train_runtime": 169.8273, + "train_samples_per_second": 2.355, + "train_steps_per_second": 0.589 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4914533793529856.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..b86edd5c0869c5a69a12bf0c6adeeda4c3b1afc2 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:386840b3ff94a28b7c83212829e6accfe75f2e328a9057cab81f329994fef0f1 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..4b425dbe03d5d986d9d1b898db1a609e55ba971e --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:99fe6a5d31215c1d0c764a74fdc4ec5918e52a0db960e55f5fdd3fe51f8dc2d2 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..22fe3ca766ec6e37d51aebf525dec2af34358ac0 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:924e12ae251af8a13eb6310668a2e9dbdc07b477f9b1ade3564b97b41171fb39 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..3199fddd167e3c1256b44534489929ba61caa48f --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:dc65366e8dea534e149ce7cd583b6319994ed35ecc39582525344b753aeb1858 +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..fa78c77c606806acc9895b6df49678c13633002e --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:92099a5ace36a30279e1cc431967135db0772dc3159d7b8d231675dfd4cf88c0 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..d9f75051c9cb79168d4847860a7429370a81c511 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:d874d9cb32ec1bddc888cfd18e4b414cf20eff538ba46baccd8b8d0b89bc4fbd +size 184221358 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..98c30135b512ee9fd0e185bb6bdfee549aa02cae --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:85cba7e8c401b0d413e0b2a269065ccf0c313caf14e5148232105b7da2d2fa8f +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..a7a1fbf8770ff4d959baf415401999d90d775e16 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b6a290cb3bf0eb2e5195b515d1ca2991a50522ba0f05ad0956cc52b0f6922957 +size 184220842 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..e84604826f0e299fa6c978a9f5b3f2eac8699cc0 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 11.693649291992188, + "learning_rate": 2e-05, + "loss": 1.0507, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 2.9311091899871826, + "learning_rate": 2e-05, + "loss": 0.4164, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 8.45428466796875, + "learning_rate": 2e-05, + "loss": 1.1758, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 14.28290843963623, + "learning_rate": 2e-05, + "loss": 1.3931, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 15.484107971191406, + "learning_rate": 2e-05, + "loss": 1.6668, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 9.924738883972168, + "learning_rate": 2e-05, + "loss": 1.5835, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 8.884476661682129, + "learning_rate": 2e-05, + "loss": 1.1145, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 9.459785461425781, + "learning_rate": 2e-05, + "loss": 0.9232, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 9.546226501464844, + "learning_rate": 2e-05, + "loss": 0.5492, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 2.949314594268799, + "learning_rate": 2e-05, + "loss": 0.744, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 5.894227027893066, + "learning_rate": 2e-05, + "loss": 0.6446, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 5.924652099609375, + "learning_rate": 2e-05, + "loss": 0.8909, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 2.6994128227233887, + "learning_rate": 2e-05, + "loss": 0.4439, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 8.177955627441406, + "learning_rate": 2e-05, + "loss": 0.4963, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 9.719780921936035, + "learning_rate": 2e-05, + "loss": 0.8525, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 22.681800842285156, + "learning_rate": 2e-05, + "loss": 1.8416, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 3.0644638538360596, + "learning_rate": 2e-05, + "loss": 2.1517, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 14.563640594482422, + "learning_rate": 2e-05, + "loss": 2.4433, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 7.03754186630249, + "learning_rate": 2e-05, + "loss": 1.3491, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 6.36046838760376, + "learning_rate": 2e-05, + "loss": 1.1212, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 18.849185943603516, + "learning_rate": 2e-05, + "loss": 1.6678, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 14.333285331726074, + "learning_rate": 2e-05, + "loss": 1.6081, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 6.0595269203186035, + "learning_rate": 2e-05, + "loss": 0.6205, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 4.74971342086792, + "learning_rate": 2e-05, + "loss": 0.4097, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 7.75990104675293, + "learning_rate": 2e-05, + "loss": 1.074, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 9.010882377624512, + "learning_rate": 2e-05, + "loss": 2.1047, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 1.1840580701828003, + "learning_rate": 2e-05, + "loss": 0.8681, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 2.5251049995422363, + "learning_rate": 2e-05, + "loss": 0.7495, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 6.284056186676025, + "learning_rate": 2e-05, + "loss": 0.8752, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 8.168758392333984, + "learning_rate": 2e-05, + "loss": 1.3925, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 6.6547770500183105, + "learning_rate": 2e-05, + "loss": 0.7414, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 7.461399078369141, + "learning_rate": 2e-05, + "loss": 0.8258, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 6.963188648223877, + "learning_rate": 2e-05, + "loss": 1.218, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 5.871964454650879, + "learning_rate": 2e-05, + "loss": 1.1124, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 7.110522747039795, + "learning_rate": 2e-05, + "loss": 1.2579, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 8.83693790435791, + "learning_rate": 2e-05, + "loss": 2.2598, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 7.1040520668029785, + "learning_rate": 2e-05, + "loss": 1.2506, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 17.06218719482422, + "learning_rate": 2e-05, + "loss": 1.4432, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 7.473016262054443, + "learning_rate": 2e-05, + "loss": 0.784, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 8.905854225158691, + "learning_rate": 2e-05, + "loss": 0.8411, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 4.500631332397461, + "learning_rate": 2e-05, + "loss": 0.2397, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 4.327226638793945, + "learning_rate": 2e-05, + "loss": 0.4464, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 4.575321674346924, + "learning_rate": 2e-05, + "loss": 1.1319, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 4.218846321105957, + "learning_rate": 2e-05, + "loss": 0.8927, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 7.213256359100342, + "learning_rate": 2e-05, + "loss": 0.9017, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 10.291433334350586, + "learning_rate": 2e-05, + "loss": 0.6296, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 6.705934524536133, + "learning_rate": 2e-05, + "loss": 0.6399, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 3.8862531185150146, + "learning_rate": 2e-05, + "loss": 0.3221, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 9.070793151855469, + "learning_rate": 2e-05, + "loss": 1.1441, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 11.29675006866455, + "learning_rate": 2e-05, + "loss": 1.2231, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 2097655350034432.0, + "train_loss": 1.0705613040924071, + "train_runtime": 102.5888, + "train_samples_per_second": 3.899, + "train_steps_per_second": 0.975 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 2097655350034432.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..ea26972b174f00ea7b91a4336211d00f1bbf6ccd --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:63c020f0e9a8f0c812148e6b98c897be1fbf16e0e6bdf9c1cb855bd2fb184967 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..85736a412801b337f711c8a1f39a4881e5956574 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:a576f6efc9cbd3bb8a501153bc3e085ec89a15902ad8cfe10bc4b974f03e4fcd +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..9f0a7334bc0ea07ab1adafbe7282923c55639e0d --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5db05f3d3601799e1fc60bea63f67d5557eee27fabad09326273c4391caa1f03 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..a1fc7bc7abc59c62ff76a6fa1523d070b0127f9e --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7b1615aeafa8813c4d2376f2ae7a14b957139e2e63d429b39c3fb322e8bde63a +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..b4660c6430d4841f8c2e176a79ac3cdd3e7b7cd9 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6ba947d6751d0af4d240a2ac8b103cec9438b64bef846a3c69fd34bf8554f5df +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..275ba55dd79e0c9eb47597a29f29c41b2c2f3084 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:56fa94a36ca7fe11a494c594eb9b640dd6ee9b47daeee27e2fee4d62575c6f43 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..5a64572f2ea255406eea5affb513586158bc3f83 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ac555e618f6ec228443d7e19fcd469b8ef8e5bea20f651ef5cde9cdf85221ad9 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..c818a8f3a31058b6182a391af6a4a66ac293729d --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9e7a27691ec0a074141418aadc3a9c16e25bbeb6a18c82bc76883e6acb92cb24 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..cce23ee7dc473a1a85953a45ba7c9ab5805ada89 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 2.637403726577759, + "learning_rate": 2e-05, + "loss": 0.5614, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 4.73873233795166, + "learning_rate": 2e-05, + "loss": 0.9697, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 1.6665256023406982, + "learning_rate": 2e-05, + "loss": 0.4683, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 3.0230000019073486, + "learning_rate": 2e-05, + "loss": 0.6475, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 2.887681245803833, + "learning_rate": 2e-05, + "loss": 0.5517, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 3.429380416870117, + "learning_rate": 2e-05, + "loss": 0.2907, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 4.610021114349365, + "learning_rate": 2e-05, + "loss": 0.6544, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 8.516341209411621, + "learning_rate": 2e-05, + "loss": 1.9958, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 3.5461740493774414, + "learning_rate": 2e-05, + "loss": 0.3917, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 5.2099456787109375, + "learning_rate": 2e-05, + "loss": 0.4288, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 4.585675239562988, + "learning_rate": 2e-05, + "loss": 1.801, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 2.1597061157226562, + "learning_rate": 2e-05, + "loss": 1.0035, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 3.1741673946380615, + "learning_rate": 2e-05, + "loss": 0.4851, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 5.476759433746338, + "learning_rate": 2e-05, + "loss": 1.1165, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 4.2731804847717285, + "learning_rate": 2e-05, + "loss": 0.9401, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 6.725061893463135, + "learning_rate": 2e-05, + "loss": 0.7967, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 2.4326884746551514, + "learning_rate": 2e-05, + "loss": 0.4525, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 4.693164348602295, + "learning_rate": 2e-05, + "loss": 1.2137, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 2.8988044261932373, + "learning_rate": 2e-05, + "loss": 0.5093, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 6.970594882965088, + "learning_rate": 2e-05, + "loss": 1.0991, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 3.4238510131835938, + "learning_rate": 2e-05, + "loss": 0.8502, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 3.9824445247650146, + "learning_rate": 2e-05, + "loss": 0.6589, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 4.8126678466796875, + "learning_rate": 2e-05, + "loss": 0.8948, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 5.100122451782227, + "learning_rate": 2e-05, + "loss": 0.8117, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 2.1359190940856934, + "learning_rate": 2e-05, + "loss": 0.4148, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 5.385488033294678, + "learning_rate": 2e-05, + "loss": 1.6419, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 4.634507656097412, + "learning_rate": 2e-05, + "loss": 0.8469, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 2.4589309692382812, + "learning_rate": 2e-05, + "loss": 0.2448, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 4.612209796905518, + "learning_rate": 2e-05, + "loss": 0.836, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 3.2502317428588867, + "learning_rate": 2e-05, + "loss": 0.8215, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 3.3926491737365723, + "learning_rate": 2e-05, + "loss": 0.6394, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 3.6580755710601807, + "learning_rate": 2e-05, + "loss": 1.1433, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 7.191780090332031, + "learning_rate": 2e-05, + "loss": 1.0154, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 2.0875039100646973, + "learning_rate": 2e-05, + "loss": 0.6591, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 7.09459114074707, + "learning_rate": 2e-05, + "loss": 0.8218, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 3.325845241546631, + "learning_rate": 2e-05, + "loss": 0.6073, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 1.3921314477920532, + "learning_rate": 2e-05, + "loss": 0.5865, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 3.6516613960266113, + "learning_rate": 2e-05, + "loss": 1.4063, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 4.696588039398193, + "learning_rate": 2e-05, + "loss": 0.7179, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 3.545356512069702, + "learning_rate": 2e-05, + "loss": 0.4085, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 3.8088479042053223, + "learning_rate": 2e-05, + "loss": 0.5266, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 5.2242326736450195, + "learning_rate": 2e-05, + "loss": 0.6016, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 2.499507427215576, + "learning_rate": 2e-05, + "loss": 0.3916, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 3.5232198238372803, + "learning_rate": 2e-05, + "loss": 0.5096, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 4.562027931213379, + "learning_rate": 2e-05, + "loss": 0.7677, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 6.415626525878906, + "learning_rate": 2e-05, + "loss": 0.7187, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 5.6967644691467285, + "learning_rate": 2e-05, + "loss": 0.7985, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 3.2610225677490234, + "learning_rate": 2e-05, + "loss": 0.6187, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 3.403749942779541, + "learning_rate": 2e-05, + "loss": 0.2896, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 11.633613586425781, + "learning_rate": 2e-05, + "loss": 1.1527, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 5694661670731776.0, + "train_loss": 0.7755919742584229, + "train_runtime": 170.0795, + "train_samples_per_second": 2.352, + "train_steps_per_second": 0.588 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5694661670731776.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..875ef31ff0602658917c89cbfc58d6319d4d4d37 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:75fce352d5052d2b03312c457b9d07a3c2045a3c1d82fa169cde815ffac140a0 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..c5932b68864f01c99fd5f5b82e6e4cb611e4240a --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:51efd3d761dbc6a85da1d36fd13afc6913e8f35c7c18b9825c9a03ebd8725119 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..b08ea826b7c2c88fd2b5c85ee58a048c4c3540b1 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04acc4730d61aa5cab2b5961a914144dea9077c80a98ea800ef7b634fb202740 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..efb36b99cc3c7422754745d8810c9c83a382d36d --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c21f2c0bf74b5e28d521eff1cae89902767d9acad0a560a77818c53885c72763 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..599b0e6863625123e3cdd051bbadd221db394fea --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:cf27c9f6c67ee9af8765ac0ea5533f11f5d1e729b6e3d53daf4b640cb3853670 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..d2ed6d0b38929d9f0b693a83ff3b6c6d651b0bca --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:f83aa707ae161bc1ddd50bd1a996c1f78a7d97983e4fdb7e4ba8894b96b33eda +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..7632e8315ae00320363c5e98e76244b18bc9cce9 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:04d4438806ea71f5e33bb20506679a54f6c0badd58b2e989c7ff596a42caa40b +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..b53da66277ada30459c7744639680cd00357a4bc --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:7d0ab48aaaf037776c5d89c9130d080f2274777ef2ae7806989abda2477652e0 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c2ad527d1c30d44d2dbb2f8bd1981fae03e072da --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 1.0338062047958374, + "learning_rate": 2e-05, + "loss": 0.7566, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 5.665828704833984, + "learning_rate": 2e-05, + "loss": 1.0194, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 6.612151145935059, + "learning_rate": 2e-05, + "loss": 1.6411, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 3.471060276031494, + "learning_rate": 2e-05, + "loss": 0.5854, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 1.2747403383255005, + "learning_rate": 2e-05, + "loss": 0.4297, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 8.470596313476562, + "learning_rate": 2e-05, + "loss": 1.0594, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 0.36429786682128906, + "learning_rate": 2e-05, + "loss": 0.3329, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 4.337061882019043, + "learning_rate": 2e-05, + "loss": 0.5275, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 5.764715671539307, + "learning_rate": 2e-05, + "loss": 1.6987, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 3.9755842685699463, + "learning_rate": 2e-05, + "loss": 1.864, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 7.792990207672119, + "learning_rate": 2e-05, + "loss": 3.8483, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 6.800496578216553, + "learning_rate": 2e-05, + "loss": 1.0842, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 5.43037223815918, + "learning_rate": 2e-05, + "loss": 1.2437, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 6.251035690307617, + "learning_rate": 2e-05, + "loss": 2.8937, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 0.5180622339248657, + "learning_rate": 2e-05, + "loss": 0.3756, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 1.7391629219055176, + "learning_rate": 2e-05, + "loss": 0.4611, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 1.569304347038269, + "learning_rate": 2e-05, + "loss": 0.215, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 6.430338382720947, + "learning_rate": 2e-05, + "loss": 1.2636, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 4.805810928344727, + "learning_rate": 2e-05, + "loss": 1.105, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 4.90680456161499, + "learning_rate": 2e-05, + "loss": 0.5404, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 3.3493094444274902, + "learning_rate": 2e-05, + "loss": 0.6253, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 5.143937110900879, + "learning_rate": 2e-05, + "loss": 0.7676, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 2.285219669342041, + "learning_rate": 2e-05, + "loss": 0.7972, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 7.806530475616455, + "learning_rate": 2e-05, + "loss": 3.4581, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 12.248115539550781, + "learning_rate": 2e-05, + "loss": 4.6998, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 5.186307907104492, + "learning_rate": 2e-05, + "loss": 2.4965, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 3.7022056579589844, + "learning_rate": 2e-05, + "loss": 0.5178, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 13.997264862060547, + "learning_rate": 2e-05, + "loss": 1.0521, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 7.620451927185059, + "learning_rate": 2e-05, + "loss": 2.9059, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 2.462949514389038, + "learning_rate": 2e-05, + "loss": 0.2984, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 3.5282416343688965, + "learning_rate": 2e-05, + "loss": 1.3804, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 7.946673393249512, + "learning_rate": 2e-05, + "loss": 1.8044, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 3.367640495300293, + "learning_rate": 2e-05, + "loss": 0.6136, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 3.411672592163086, + "learning_rate": 2e-05, + "loss": 0.8734, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 6.9260969161987305, + "learning_rate": 2e-05, + "loss": 3.405, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 2.8345489501953125, + "learning_rate": 2e-05, + "loss": 2.5527, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 0.7165583372116089, + "learning_rate": 2e-05, + "loss": 0.4615, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 8.22752857208252, + "learning_rate": 2e-05, + "loss": 0.9082, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 4.878330707550049, + "learning_rate": 2e-05, + "loss": 1.7125, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 6.852156639099121, + "learning_rate": 2e-05, + "loss": 1.0581, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 4.600001335144043, + "learning_rate": 2e-05, + "loss": 0.8398, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 4.03223991394043, + "learning_rate": 2e-05, + "loss": 0.7341, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 2.608586072921753, + "learning_rate": 2e-05, + "loss": 0.3577, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 7.7058424949646, + "learning_rate": 2e-05, + "loss": 0.9836, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 2.9475879669189453, + "learning_rate": 2e-05, + "loss": 0.3618, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 7.979467391967773, + "learning_rate": 2e-05, + "loss": 1.1729, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 5.021623611450195, + "learning_rate": 2e-05, + "loss": 0.5304, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 10.417924880981445, + "learning_rate": 2e-05, + "loss": 5.0102, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 0.491908997297287, + "learning_rate": 2e-05, + "loss": 2.575, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 7.822747707366943, + "learning_rate": 2e-05, + "loss": 1.7126, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 4901658932805632.0, + "train_loss": 1.3922410774230958, + "train_runtime": 170.2011, + "train_samples_per_second": 2.35, + "train_steps_per_second": 0.588 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 4901658932805632.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..450048d95703dc653eb333567ab2f9cd68d07eb6 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:96f6f7356fb9a5a9dd2a32d459145266c4872f8dafddd2a1b42c364247ec8d1c +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..a6fbe39e373bdbfd107dae3782b7893703b21626 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:25c6be73f4fc66311d77d70650419c2a97c8790f3c3d84f647af59dc02768e4e +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..3397f65713a1de89bca9d576095b085df121e608 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:c81e7e0ad944248aa4f4b98e41afbd836df459feca97c8fb9707bd6b80b71ce1 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..89ff0dd61545afe45726fbee8c2c99b3427f4949 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ec0a8f4ff9daea9e7beb8c98c1b0ec4d9095c883738a1088014aaa0a7bf48f8a +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..12160a9a529958da92318811e514ce6b95c06abe --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:98715cd26748066462045c4642894e1027030461bb77ad87770cc76280c54a3b +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..05982cc8fe8038f2f156890af8c33a408ad19a64 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:b58f5e3640f9bff8535d561cccaedb6f3ae0d6ab6e7ec2f903b0fe5476b7d5a2 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..fe85a8db8d0505b352d5733159222903fa505b85 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:853c02dd7bd55df85125d54791c6918907a6fdf90f4ad1da31f424ec046d79b8 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..2658e5557745e538a528e488844cab647ce89913 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:6b1a1d15929d7074e34afe26791a162fde4f54760b447fa045085f2bb515bfce +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..c5e5671291f6d8348dc478dff6c19229dd636c87 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 6.350866794586182, + "learning_rate": 2e-05, + "loss": 0.8442, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 7.095775604248047, + "learning_rate": 2e-05, + "loss": 0.862, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 6.973437786102295, + "learning_rate": 2e-05, + "loss": 1.0801, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 2.7381036281585693, + "learning_rate": 2e-05, + "loss": 0.8341, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 5.858903408050537, + "learning_rate": 2e-05, + "loss": 1.3088, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 1.146891474723816, + "learning_rate": 2e-05, + "loss": 0.1271, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 5.8700642585754395, + "learning_rate": 2e-05, + "loss": 0.6333, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 10.154126167297363, + "learning_rate": 2e-05, + "loss": 1.3238, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 7.10991096496582, + "learning_rate": 2e-05, + "loss": 1.7668, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 4.042394161224365, + "learning_rate": 2e-05, + "loss": 0.7352, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 4.039437770843506, + "learning_rate": 2e-05, + "loss": 0.792, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 1.5868761539459229, + "learning_rate": 2e-05, + "loss": 0.7132, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 2.3178021907806396, + "learning_rate": 2e-05, + "loss": 0.7587, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 6.503546237945557, + "learning_rate": 2e-05, + "loss": 0.8239, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 6.615965366363525, + "learning_rate": 2e-05, + "loss": 1.5844, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 6.158043384552002, + "learning_rate": 2e-05, + "loss": 0.7461, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 3.276494026184082, + "learning_rate": 2e-05, + "loss": 0.9598, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 3.5142297744750977, + "learning_rate": 2e-05, + "loss": 1.0434, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 2.0729358196258545, + "learning_rate": 2e-05, + "loss": 0.8575, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 3.127547264099121, + "learning_rate": 2e-05, + "loss": 0.8279, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 3.0067636966705322, + "learning_rate": 2e-05, + "loss": 1.3029, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 4.825779914855957, + "learning_rate": 2e-05, + "loss": 0.5708, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 5.970290184020996, + "learning_rate": 2e-05, + "loss": 0.7744, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 5.233969211578369, + "learning_rate": 2e-05, + "loss": 0.7049, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 3.52846622467041, + "learning_rate": 2e-05, + "loss": 0.9376, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 1.7202891111373901, + "learning_rate": 2e-05, + "loss": 0.749, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 4.498488903045654, + "learning_rate": 2e-05, + "loss": 0.7168, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 5.07948112487793, + "learning_rate": 2e-05, + "loss": 1.7681, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 7.285374164581299, + "learning_rate": 2e-05, + "loss": 0.9782, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 3.47748064994812, + "learning_rate": 2e-05, + "loss": 1.2967, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 5.545077800750732, + "learning_rate": 2e-05, + "loss": 0.9142, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 2.9953062534332275, + "learning_rate": 2e-05, + "loss": 0.8001, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 2.4720499515533447, + "learning_rate": 2e-05, + "loss": 1.0479, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 6.32978630065918, + "learning_rate": 2e-05, + "loss": 1.0165, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 3.1678273677825928, + "learning_rate": 2e-05, + "loss": 1.0068, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 1.8918789625167847, + "learning_rate": 2e-05, + "loss": 0.6592, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 7.592050075531006, + "learning_rate": 2e-05, + "loss": 1.2711, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 4.536243915557861, + "learning_rate": 2e-05, + "loss": 1.1003, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 2.779127836227417, + "learning_rate": 2e-05, + "loss": 0.494, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 3.7273905277252197, + "learning_rate": 2e-05, + "loss": 0.4256, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 3.0089080333709717, + "learning_rate": 2e-05, + "loss": 0.8304, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 3.3422775268554688, + "learning_rate": 2e-05, + "loss": 1.3364, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 4.476714134216309, + "learning_rate": 2e-05, + "loss": 0.4442, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 2.4386043548583984, + "learning_rate": 2e-05, + "loss": 0.8881, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 6.396787166595459, + "learning_rate": 2e-05, + "loss": 0.5421, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 2.9071009159088135, + "learning_rate": 2e-05, + "loss": 2.2692, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 2.4817564487457275, + "learning_rate": 2e-05, + "loss": 0.5462, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 10.91292953491211, + "learning_rate": 2e-05, + "loss": 1.1156, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 7.508020401000977, + "learning_rate": 2e-05, + "loss": 1.2463, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 7.0632500648498535, + "learning_rate": 2e-05, + "loss": 1.967, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 9847400269086720.0, + "train_loss": 0.9668588447570801, + "train_runtime": 198.3477, + "train_samples_per_second": 2.017, + "train_steps_per_second": 0.504 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 9847400269086720.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..ecd3324a272ad2e19c7c5beb80029382e5e9281a --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:78b485308edf5a924f2ef57a05c65017aa735f4f8f751d6d24147be61a379cdd +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..d3cef4e2b408a96fa76cf196961b568845bea1db --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:13a51c7a63f8ddcbe3f0878625d00b95735ba43292babc2a0968ae38b3222b8f +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..4402ece2c150a460dfb18019ce1b7e10c7e2155d --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:15f95e115c2ab8cc0417b6de7b16011a72e457dd83bff996f6e0595d94aca276 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..3089429fa4b6f8ec7c27c19c37481eaabb9c575f --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:ce54c06b61b3bb1eb86188b4a03197e6d959d69b5cc7b6e72d2969d352596b74 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..48f2225e8df0876ec24175e4680df9ce285f7e69 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:74adb04a0f7c2f9bd64a9979982262c86e0a658cdb2a126b4d1af7db859602f0 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..ac7154657cde3cbfe9ea0745960d39c5fa2bbd64 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:27a43238a35228ee66d44ddf92b40183b15124a43d59e49640cb47ed81a86b68 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..a70fd99732154db654a6694c9026e74e127e7a48 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:9b078ac3ebb4966747b773b17e3a0b4a019db00d3e099deba6c28f83ce3a7257 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..9d1d167d0bf4481186a8c85c09d03169d8f0034a --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:0b9e95bcd67b3d6f50550885ae19aac23a3a1a33c664205e2ac601aa0367f43c +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..b493ce4acbee516d93764e564063ce12eead09e4 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 2.5717689990997314, + "learning_rate": 2e-05, + "loss": 0.1378, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 6.968014717102051, + "learning_rate": 2e-05, + "loss": 0.7146, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 4.446895599365234, + "learning_rate": 2e-05, + "loss": 0.3686, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 1.4856871366500854, + "learning_rate": 2e-05, + "loss": 0.5728, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 4.265855312347412, + "learning_rate": 2e-05, + "loss": 1.149, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 0.34307774901390076, + "learning_rate": 2e-05, + "loss": 0.1337, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 0.6195315718650818, + "learning_rate": 2e-05, + "loss": 0.1026, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 3.3206229209899902, + "learning_rate": 2e-05, + "loss": 0.8001, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 4.18081521987915, + "learning_rate": 2e-05, + "loss": 0.3459, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 6.578738689422607, + "learning_rate": 2e-05, + "loss": 0.4957, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 1.6412115097045898, + "learning_rate": 2e-05, + "loss": 0.3151, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 5.7735209465026855, + "learning_rate": 2e-05, + "loss": 0.4686, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 3.0710041522979736, + "learning_rate": 2e-05, + "loss": 0.1932, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 2.2720937728881836, + "learning_rate": 2e-05, + "loss": 0.3962, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 0.873317301273346, + "learning_rate": 2e-05, + "loss": 1.1774, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 2.1178581714630127, + "learning_rate": 2e-05, + "loss": 0.5692, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 4.69739294052124, + "learning_rate": 2e-05, + "loss": 0.2105, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 5.956394195556641, + "learning_rate": 2e-05, + "loss": 1.0291, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 0.9587688446044922, + "learning_rate": 2e-05, + "loss": 0.1533, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 9.836400985717773, + "learning_rate": 2e-05, + "loss": 1.6179, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 0.8551999926567078, + "learning_rate": 2e-05, + "loss": 0.7573, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 4.6954827308654785, + "learning_rate": 2e-05, + "loss": 0.2751, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 3.2609949111938477, + "learning_rate": 2e-05, + "loss": 0.699, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 0.6860560774803162, + "learning_rate": 2e-05, + "loss": 0.0301, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 0.28854402899742126, + "learning_rate": 2e-05, + "loss": 1.0212, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 4.731658935546875, + "learning_rate": 2e-05, + "loss": 1.2925, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 0.11632199585437775, + "learning_rate": 2e-05, + "loss": 0.8402, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 0.5225258469581604, + "learning_rate": 2e-05, + "loss": 0.047, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 0.5997475981712341, + "learning_rate": 2e-05, + "loss": 0.3446, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 10.83228588104248, + "learning_rate": 2e-05, + "loss": 2.1025, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 0.7379941940307617, + "learning_rate": 2e-05, + "loss": 0.0965, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 5.670101642608643, + "learning_rate": 2e-05, + "loss": 0.9854, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 1.83771550655365, + "learning_rate": 2e-05, + "loss": 0.2185, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 2.517608642578125, + "learning_rate": 2e-05, + "loss": 0.2315, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 6.244460582733154, + "learning_rate": 2e-05, + "loss": 0.9419, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 4.1970391273498535, + "learning_rate": 2e-05, + "loss": 0.6381, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 7.284065246582031, + "learning_rate": 2e-05, + "loss": 1.0512, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 1.3797696828842163, + "learning_rate": 2e-05, + "loss": 0.7621, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 1.6824308633804321, + "learning_rate": 2e-05, + "loss": 0.7541, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 8.63947582244873, + "learning_rate": 2e-05, + "loss": 1.9929, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 2.2472915649414062, + "learning_rate": 2e-05, + "loss": 0.2506, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 3.2342183589935303, + "learning_rate": 2e-05, + "loss": 0.3462, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 3.877340078353882, + "learning_rate": 2e-05, + "loss": 1.3992, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 7.802109718322754, + "learning_rate": 2e-05, + "loss": 1.2399, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 3.077086925506592, + "learning_rate": 2e-05, + "loss": 1.1619, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 0.8278634548187256, + "learning_rate": 2e-05, + "loss": 0.3156, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 1.945942997932434, + "learning_rate": 2e-05, + "loss": 0.1714, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 0.2935602366924286, + "learning_rate": 2e-05, + "loss": 0.2344, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 2.8402061462402344, + "learning_rate": 2e-05, + "loss": 0.3634, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 5.449839115142822, + "learning_rate": 2e-05, + "loss": 0.455, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 5158236269314048.0, + "train_loss": 0.6394131851196289, + "train_runtime": 170.8215, + "train_samples_per_second": 2.342, + "train_steps_per_second": 0.585 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5158236269314048.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +} diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round10.pth new file mode 100644 index 0000000000000000000000000000000000000000..e1561a8285a85251eaba0aa5e884a8b94d9ab767 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round10.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:5741134779b229b448c5e3fc8024d44bffdbd9ebca7fdcc3bd880aa835a31280 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round12.pth new file mode 100644 index 0000000000000000000000000000000000000000..7e506cf512eeb98a8e3b0b437428c61677218065 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round12.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:2b374235edfcd56ee363e78bcea30bc6b0218b6c63c9bc3ddb516e6efcbe64eb +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round15.pth new file mode 100644 index 0000000000000000000000000000000000000000..3b8f5fc68d3bb62d79ecf1de1a1452f773780fe1 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round15.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:01f20d696391c4300f109f34470924e22885d200aed6b1c45cd2b11e3f60b61b +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round17.pth new file mode 100644 index 0000000000000000000000000000000000000000..32227b60a637028df36cc93f527e712e55a021f7 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round17.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:47e49964d10a8d173547a0ce060575e8d0dc9a9b431f321f8aacd61a08e785a7 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round2.pth new file mode 100644 index 0000000000000000000000000000000000000000..3215023ab2eb1a85d819238dfaaa69910010fd11 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round2.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:025f4c77a9baea1bebbef6e42fff1c398590fb24c6e55c197cb55b50e94dd682 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round20.pth new file mode 100644 index 0000000000000000000000000000000000000000..36de5320d0fc9c1cda83892249b0488e0cc53256 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round20.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:1abbefa222ae9042b4bb67523aec2c06a4b0917fe3bb41469a8bce791ea0cc84 +size 395787774 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round5.pth new file mode 100644 index 0000000000000000000000000000000000000000..49253fdce3d90d924c117c0b02fbe4fb3cbaa1f0 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round5.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:e4eb8227422e278d8dfabde57874b10b62347351bfd9185e17b97be3ea560c38 +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round7.pth new file mode 100644 index 0000000000000000000000000000000000000000..1a31f364d5a0b57e83c028d01ada716299d30203 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round7.pth @@ -0,0 +1,3 @@ +version https://git-lfs.github.com/spec/v1 +oid sha256:79cc6353eb47ef09f9cdd8d06935408c221f92ffcea3c01c2bb67086e201bbde +size 395786922 diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_trainer_state.json new file mode 100644 index 0000000000000000000000000000000000000000..aa8b73f3ee317fdfcfc055ba5d42addebc5912c8 --- /dev/null +++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_trainer_state.json @@ -0,0 +1,392 @@ +{ + "best_metric": null, + "best_model_checkpoint": null, + "epoch": 1.0, + "eval_steps": 500, + "global_step": 100, + "is_hyper_param_search": false, + "is_local_process_zero": true, + "is_world_process_zero": true, + "log_history": [ + { + "epoch": 0.02, + "grad_norm": 0.16113965213298798, + "learning_rate": 2e-05, + "loss": 0.0915, + "step": 2 + }, + { + "epoch": 0.04, + "grad_norm": 0.07829011231660843, + "learning_rate": 2e-05, + "loss": 0.1069, + "step": 4 + }, + { + "epoch": 0.06, + "grad_norm": 0.37092337012290955, + "learning_rate": 2e-05, + "loss": 0.131, + "step": 6 + }, + { + "epoch": 0.08, + "grad_norm": 3.931626796722412, + "learning_rate": 2e-05, + "loss": 0.4551, + "step": 8 + }, + { + "epoch": 0.1, + "grad_norm": 0.05373501405119896, + "learning_rate": 2e-05, + "loss": 0.0031, + "step": 10 + }, + { + "epoch": 0.12, + "grad_norm": 2.1131927967071533, + "learning_rate": 2e-05, + "loss": 0.1581, + "step": 12 + }, + { + "epoch": 0.14, + "grad_norm": 0.07835656404495239, + "learning_rate": 2e-05, + "loss": 0.3279, + "step": 14 + }, + { + "epoch": 0.16, + "grad_norm": 8.693585395812988, + "learning_rate": 2e-05, + "loss": 0.5947, + "step": 16 + }, + { + "epoch": 0.18, + "grad_norm": 2.7874717712402344, + "learning_rate": 2e-05, + "loss": 0.2964, + "step": 18 + }, + { + "epoch": 0.2, + "grad_norm": 6.932012557983398, + "learning_rate": 2e-05, + "loss": 3.8206, + "step": 20 + }, + { + "epoch": 0.22, + "grad_norm": 5.361955642700195, + "learning_rate": 2e-05, + "loss": 0.7578, + "step": 22 + }, + { + "epoch": 0.24, + "grad_norm": 0.567628800868988, + "learning_rate": 2e-05, + "loss": 0.9104, + "step": 24 + }, + { + "epoch": 0.26, + "grad_norm": 2.3103795051574707, + "learning_rate": 2e-05, + "loss": 0.157, + "step": 26 + }, + { + "epoch": 0.28, + "grad_norm": 1.790549874305725, + "learning_rate": 2e-05, + "loss": 0.1, + "step": 28 + }, + { + "epoch": 0.3, + "grad_norm": 2.410795211791992, + "learning_rate": 2e-05, + "loss": 0.1141, + "step": 30 + }, + { + "epoch": 0.32, + "grad_norm": 0.02548924647271633, + "learning_rate": 2e-05, + "loss": 0.4958, + "step": 32 + }, + { + "epoch": 0.34, + "grad_norm": 4.374381065368652, + "learning_rate": 2e-05, + "loss": 1.8172, + "step": 34 + }, + { + "epoch": 0.36, + "grad_norm": 3.1204335689544678, + "learning_rate": 2e-05, + "loss": 0.6653, + "step": 36 + }, + { + "epoch": 0.38, + "grad_norm": 4.962135314941406, + "learning_rate": 2e-05, + "loss": 0.7358, + "step": 38 + }, + { + "epoch": 0.4, + "grad_norm": 2.9371933937072754, + "learning_rate": 2e-05, + "loss": 1.0185, + "step": 40 + }, + { + "epoch": 0.42, + "grad_norm": 6.699105739593506, + "learning_rate": 2e-05, + "loss": 0.4942, + "step": 42 + }, + { + "epoch": 0.44, + "grad_norm": 0.01869381032884121, + "learning_rate": 2e-05, + "loss": 0.0809, + "step": 44 + }, + { + "epoch": 0.46, + "grad_norm": 5.275974273681641, + "learning_rate": 2e-05, + "loss": 0.8099, + "step": 46 + }, + { + "epoch": 0.48, + "grad_norm": 6.025607109069824, + "learning_rate": 2e-05, + "loss": 1.2858, + "step": 48 + }, + { + "epoch": 0.5, + "grad_norm": 0.7212814688682556, + "learning_rate": 2e-05, + "loss": 0.4049, + "step": 50 + }, + { + "epoch": 0.52, + "grad_norm": 0.0265222005546093, + "learning_rate": 2e-05, + "loss": 0.4391, + "step": 52 + }, + { + "epoch": 0.54, + "grad_norm": 0.3079047203063965, + "learning_rate": 2e-05, + "loss": 0.7052, + "step": 54 + }, + { + "epoch": 0.56, + "grad_norm": 0.7660133838653564, + "learning_rate": 2e-05, + "loss": 0.4831, + "step": 56 + }, + { + "epoch": 0.58, + "grad_norm": 2.1217522621154785, + "learning_rate": 2e-05, + "loss": 0.1372, + "step": 58 + }, + { + "epoch": 0.6, + "grad_norm": 0.07789596170186996, + "learning_rate": 2e-05, + "loss": 0.012, + "step": 60 + }, + { + "epoch": 0.62, + "grad_norm": 6.876741409301758, + "learning_rate": 2e-05, + "loss": 0.4723, + "step": 62 + }, + { + "epoch": 0.64, + "grad_norm": 8.655380249023438, + "learning_rate": 2e-05, + "loss": 1.6293, + "step": 64 + }, + { + "epoch": 0.66, + "grad_norm": 3.3928043842315674, + "learning_rate": 2e-05, + "loss": 0.2042, + "step": 66 + }, + { + "epoch": 0.68, + "grad_norm": 0.3449620306491852, + "learning_rate": 2e-05, + "loss": 0.0418, + "step": 68 + }, + { + "epoch": 0.7, + "grad_norm": 0.8447442650794983, + "learning_rate": 2e-05, + "loss": 0.0398, + "step": 70 + }, + { + "epoch": 0.72, + "grad_norm": 5.244335651397705, + "learning_rate": 2e-05, + "loss": 0.3679, + "step": 72 + }, + { + "epoch": 0.74, + "grad_norm": 1.5702224969863892, + "learning_rate": 2e-05, + "loss": 0.0593, + "step": 74 + }, + { + "epoch": 0.76, + "grad_norm": 1.0168324708938599, + "learning_rate": 2e-05, + "loss": 0.0725, + "step": 76 + }, + { + "epoch": 0.78, + "grad_norm": 0.3886169493198395, + "learning_rate": 2e-05, + "loss": 0.2775, + "step": 78 + }, + { + "epoch": 0.8, + "grad_norm": 0.5766316652297974, + "learning_rate": 2e-05, + "loss": 0.0425, + "step": 80 + }, + { + "epoch": 0.82, + "grad_norm": 7.655923843383789, + "learning_rate": 2e-05, + "loss": 1.0432, + "step": 82 + }, + { + "epoch": 0.84, + "grad_norm": 1.8649804592132568, + "learning_rate": 2e-05, + "loss": 0.0783, + "step": 84 + }, + { + "epoch": 0.86, + "grad_norm": 0.19814282655715942, + "learning_rate": 2e-05, + "loss": 0.0133, + "step": 86 + }, + { + "epoch": 0.88, + "grad_norm": 2.5453948974609375, + "learning_rate": 2e-05, + "loss": 0.1318, + "step": 88 + }, + { + "epoch": 0.9, + "grad_norm": 1.7596789598464966, + "learning_rate": 2e-05, + "loss": 0.544, + "step": 90 + }, + { + "epoch": 0.92, + "grad_norm": 8.948413848876953, + "learning_rate": 2e-05, + "loss": 1.3617, + "step": 92 + }, + { + "epoch": 0.94, + "grad_norm": 0.24089735746383667, + "learning_rate": 2e-05, + "loss": 0.0106, + "step": 94 + }, + { + "epoch": 0.96, + "grad_norm": 1.874826192855835, + "learning_rate": 2e-05, + "loss": 0.2241, + "step": 96 + }, + { + "epoch": 0.98, + "grad_norm": 0.6936571598052979, + "learning_rate": 2e-05, + "loss": 0.0262, + "step": 98 + }, + { + "epoch": 1.0, + "grad_norm": 0.33103814721107483, + "learning_rate": 2e-05, + "loss": 0.0278, + "step": 100 + }, + { + "epoch": 1.0, + "step": 100, + "total_flos": 5020151544020992.0, + "train_loss": 0.48555011510849, + "train_runtime": 169.5382, + "train_samples_per_second": 2.359, + "train_steps_per_second": 0.59 + } + ], + "logging_steps": 2, + "max_steps": 100, + "num_input_tokens_seen": 0, + "num_train_epochs": 1, + "save_steps": 500, + "stateful_callbacks": { + "TrainerControl": { + "args": { + "should_epoch_stop": false, + "should_evaluate": false, + "should_log": false, + "should_save": false, + "should_training_stop": false + }, + "attributes": {} + } + }, + "total_flos": 5020151544020992.0, + "train_batch_size": 1, + "trial_name": null, + "trial_params": null +}