diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..7d870d5d5532ee64327c74b3ae4625aca1fbc678
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0d45b60c3377d2b233fb79150e4c6377564419264b3e0dd37b143f4dea3eeaca
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..ae9d1dd9bd9f40d28570f213271e02c8007884ef
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d914723f4cff80df3e933a3d63ed043f08536b104877dd055d63014f403de212
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..067666c96d57647af15b9172a0b8c14d0a18f5cf
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:af842a7f9e95aa8f8e6a126bead7c15243b1d3dd91e7dd5062278c436136c8d6
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..23b3d46da14fee4b9971453aa641c15f647582a4
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b7608843e5153961ba8c3edc1f197c34164f0867a809f0207e1008fecf20a06a
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..fb7a1c55b2239d507664c916b5f5c3d1081436d4
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:44243ca3716547c0c21a27577345ceaeb9e623be86ced11e1044d7ab6fa8a7e7
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..54fbdcfef68daf299c1a7a4c95ca9d3d2f2aef36
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ea2112c3636246412e69c2fcd14b814cd7b68334b9c3f8bbddb5b8208429e04a
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..6ba4e4d50e853148e9d60ef65243953ed03f963c
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1b8c7efe671d91efab26d3fbdc245783660112915cbb1d22a44ed154e94e2748
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..beba9ed658e3e1a25a91f08ff25db64d1906e81a
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:291f338c17fa2a84cb07d5db8619e1333d6e2f8944c902a3261c101b354ba731
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..97d2391e209c550ce4b08aadac16ff6a66d3ef96
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 10.044001579284668,
+      "learning_rate": 2e-05,
+      "loss": 1.0811,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 5.708449363708496,
+      "learning_rate": 2e-05,
+      "loss": 1.0935,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 6.164332389831543,
+      "learning_rate": 2e-05,
+      "loss": 0.2887,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 0.7329395413398743,
+      "learning_rate": 2e-05,
+      "loss": 0.0654,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 3.7981791496276855,
+      "learning_rate": 2e-05,
+      "loss": 0.5255,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 10.993570327758789,
+      "learning_rate": 2e-05,
+      "loss": 0.7257,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 6.135924816131592,
+      "learning_rate": 2e-05,
+      "loss": 0.3716,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 10.345540046691895,
+      "learning_rate": 2e-05,
+      "loss": 1.132,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 15.495490074157715,
+      "learning_rate": 2e-05,
+      "loss": 1.7626,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 9.968962669372559,
+      "learning_rate": 2e-05,
+      "loss": 1.0484,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 0.6021881103515625,
+      "learning_rate": 2e-05,
+      "loss": 0.0195,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 10.388916015625,
+      "learning_rate": 2e-05,
+      "loss": 0.813,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 6.319398880004883,
+      "learning_rate": 2e-05,
+      "loss": 0.2256,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 0.6123639345169067,
+      "learning_rate": 2e-05,
+      "loss": 0.1623,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 2.0997838973999023,
+      "learning_rate": 2e-05,
+      "loss": 0.8552,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 17.500144958496094,
+      "learning_rate": 2e-05,
+      "loss": 1.1279,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 5.181461334228516,
+      "learning_rate": 2e-05,
+      "loss": 0.2672,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 10.427047729492188,
+      "learning_rate": 2e-05,
+      "loss": 0.5116,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 4.001834392547607,
+      "learning_rate": 2e-05,
+      "loss": 0.8254,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 4.918587684631348,
+      "learning_rate": 2e-05,
+      "loss": 0.2577,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 1.5070538520812988,
+      "learning_rate": 2e-05,
+      "loss": 0.0786,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 8.23068904876709,
+      "learning_rate": 2e-05,
+      "loss": 1.336,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 0.4960165321826935,
+      "learning_rate": 2e-05,
+      "loss": 0.2786,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 14.814494132995605,
+      "learning_rate": 2e-05,
+      "loss": 1.3887,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 2.1397197246551514,
+      "learning_rate": 2e-05,
+      "loss": 0.1331,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 11.526412963867188,
+      "learning_rate": 2e-05,
+      "loss": 0.4426,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 2.414958953857422,
+      "learning_rate": 2e-05,
+      "loss": 0.2413,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 4.718291282653809,
+      "learning_rate": 2e-05,
+      "loss": 0.1895,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 12.340472221374512,
+      "learning_rate": 2e-05,
+      "loss": 1.9869,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 4.813292503356934,
+      "learning_rate": 2e-05,
+      "loss": 0.1454,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 2.3212244510650635,
+      "learning_rate": 2e-05,
+      "loss": 0.4304,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 3.7471375465393066,
+      "learning_rate": 2e-05,
+      "loss": 0.1793,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 0.6779154539108276,
+      "learning_rate": 2e-05,
+      "loss": 0.1912,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 5.2416229248046875,
+      "learning_rate": 2e-05,
+      "loss": 0.7198,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 12.138299942016602,
+      "learning_rate": 2e-05,
+      "loss": 0.6979,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 8.756268501281738,
+      "learning_rate": 2e-05,
+      "loss": 0.3211,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 12.594830513000488,
+      "learning_rate": 2e-05,
+      "loss": 2.7682,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 5.925561904907227,
+      "learning_rate": 2e-05,
+      "loss": 0.7106,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 0.4183579981327057,
+      "learning_rate": 2e-05,
+      "loss": 0.9514,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 6.107239246368408,
+      "learning_rate": 2e-05,
+      "loss": 1.1018,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 6.866254806518555,
+      "learning_rate": 2e-05,
+      "loss": 0.6412,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 13.476661682128906,
+      "learning_rate": 2e-05,
+      "loss": 0.9713,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 0.8483925461769104,
+      "learning_rate": 2e-05,
+      "loss": 0.1767,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 9.308485984802246,
+      "learning_rate": 2e-05,
+      "loss": 1.2229,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 4.227640151977539,
+      "learning_rate": 2e-05,
+      "loss": 0.4012,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 0.6228331923484802,
+      "learning_rate": 2e-05,
+      "loss": 0.9619,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 9.9182767868042,
+      "learning_rate": 2e-05,
+      "loss": 2.1173,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 3.4465816020965576,
+      "learning_rate": 2e-05,
+      "loss": 1.605,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 12.18694019317627,
+      "learning_rate": 2e-05,
+      "loss": 2.0198,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 1.0387413501739502,
+      "learning_rate": 2e-05,
+      "loss": 0.2066,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 2053257199353856.0,
+      "train_loss": 0.755521228313446,
+      "train_runtime": 101.7028,
+      "train_samples_per_second": 3.933,
+      "train_steps_per_second": 0.983
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 2053257199353856.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a3e7f1bdc039c9103df345744bb709a2c9edb513
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:bf0af71fd4d6dc12cffd01523df4bbc58a571c0978ca4c7d7571aa24bdeb8ee5
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..86ef678a775e7ca28a99403a265f1afc13cacb1a
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1744e2ae1501079d9883e05d012fe5bf1d7a4f72884671f31cc3a124d304ec02
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..715a90d3af60602b1aa6683fa2d537e31759956f
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c105cc5e4f96671ece69b8f9d708aa9a805d9a65a48b0865cc54c7d7fc57fdbf
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..db1f81c5bdea0994ca69f0beb95a6d6afc777060
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ff0a15058359453e6d079b47f6e618fcf9720db11d691877dbe2140c6a15be3c
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a4f2b2deb6a64520937f91b370638b8fe39c5e9d
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1ec3b7d6ad6a720fb656550c2f62665f56860a6f8a8288ff1f826b9dceaed89e
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..dd75619c4e77c7d16e4fe5540d316a29dc209386
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:59993217d193880b0fc5548c04f04df882e9949136543c7bbede6e4dcfb2687c
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..6c4811c41c95c876a6bab828a837f05f15d603a0
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:009defbeb3403fee4f1ed358ade2be16ece2a7c97a4f95464240ea4cf6c2e3d8
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..c2dc6c8bf124f8d967dd0e1c48bed05e6470295d
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c0be9e6109694003abb07f24a9bb7b03b63669d481039c2e19e40b6296e98501
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..3f538efe59f01f293b553ba218b2212bf0d09ac5
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 0.02960115671157837,
+      "learning_rate": 2e-05,
+      "loss": 0.0045,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 0.09186350554227829,
+      "learning_rate": 2e-05,
+      "loss": 0.0653,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 0.16298067569732666,
+      "learning_rate": 2e-05,
+      "loss": 0.0063,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 0.09170705080032349,
+      "learning_rate": 2e-05,
+      "loss": 0.0034,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 0.06864725798368454,
+      "learning_rate": 2e-05,
+      "loss": 0.0031,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 0.19071757793426514,
+      "learning_rate": 2e-05,
+      "loss": 0.0287,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 0.17343077063560486,
+      "learning_rate": 2e-05,
+      "loss": 0.0038,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 0.2856125235557556,
+      "learning_rate": 2e-05,
+      "loss": 0.0591,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 0.04609253257513046,
+      "learning_rate": 2e-05,
+      "loss": 0.024,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 0.01641538180410862,
+      "learning_rate": 2e-05,
+      "loss": 0.001,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 0.07860966771841049,
+      "learning_rate": 2e-05,
+      "loss": 0.0113,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 0.027962295338511467,
+      "learning_rate": 2e-05,
+      "loss": 0.0011,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 1.3819764852523804,
+      "learning_rate": 2e-05,
+      "loss": 0.0356,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 0.01595745049417019,
+      "learning_rate": 2e-05,
+      "loss": 0.0008,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 0.12123987823724747,
+      "learning_rate": 2e-05,
+      "loss": 0.004,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 0.7385960221290588,
+      "learning_rate": 2e-05,
+      "loss": 0.1151,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 5.604942798614502,
+      "learning_rate": 2e-05,
+      "loss": 0.1588,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 0.005619928240776062,
+      "learning_rate": 2e-05,
+      "loss": 0.0003,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 0.027213625609874725,
+      "learning_rate": 2e-05,
+      "loss": 0.0016,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 4.135501384735107,
+      "learning_rate": 2e-05,
+      "loss": 0.0975,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 0.00542866624891758,
+      "learning_rate": 2e-05,
+      "loss": 0.0006,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 7.716265678405762,
+      "learning_rate": 2e-05,
+      "loss": 0.3077,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 0.01432905811816454,
+      "learning_rate": 2e-05,
+      "loss": 0.0009,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 0.3185250163078308,
+      "learning_rate": 2e-05,
+      "loss": 0.004,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 0.00824733730405569,
+      "learning_rate": 2e-05,
+      "loss": 0.0022,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 0.006523216143250465,
+      "learning_rate": 2e-05,
+      "loss": 0.0265,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 0.3274787664413452,
+      "learning_rate": 2e-05,
+      "loss": 0.0094,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 0.012340985238552094,
+      "learning_rate": 2e-05,
+      "loss": 0.0011,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 0.01708126626908779,
+      "learning_rate": 2e-05,
+      "loss": 0.6785,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 0.005370909348130226,
+      "learning_rate": 2e-05,
+      "loss": 0.2075,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 0.009917240589857101,
+      "learning_rate": 2e-05,
+      "loss": 0.0011,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 0.015617340803146362,
+      "learning_rate": 2e-05,
+      "loss": 0.0178,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 0.04841961711645126,
+      "learning_rate": 2e-05,
+      "loss": 0.0013,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 2.055001735687256,
+      "learning_rate": 2e-05,
+      "loss": 0.0487,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 0.010724334977567196,
+      "learning_rate": 2e-05,
+      "loss": 0.0007,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 0.1112007200717926,
+      "learning_rate": 2e-05,
+      "loss": 0.0032,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 0.5635241270065308,
+      "learning_rate": 2e-05,
+      "loss": 0.0103,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 0.024422986432909966,
+      "learning_rate": 2e-05,
+      "loss": 0.0009,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 9.877224922180176,
+      "learning_rate": 2e-05,
+      "loss": 0.4788,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 0.025433897972106934,
+      "learning_rate": 2e-05,
+      "loss": 0.0017,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 0.07279513031244278,
+      "learning_rate": 2e-05,
+      "loss": 0.0021,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 0.003639621427282691,
+      "learning_rate": 2e-05,
+      "loss": 0.0004,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 0.021531013771891594,
+      "learning_rate": 2e-05,
+      "loss": 1.1024,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 0.041703518480062485,
+      "learning_rate": 2e-05,
+      "loss": 0.0053,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 1.0306504964828491,
+      "learning_rate": 2e-05,
+      "loss": 0.0258,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 0.19764401018619537,
+      "learning_rate": 2e-05,
+      "loss": 0.0066,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 0.1398591846227646,
+      "learning_rate": 2e-05,
+      "loss": 0.0036,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 0.9593293070793152,
+      "learning_rate": 2e-05,
+      "loss": 0.0234,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 0.013573438860476017,
+      "learning_rate": 2e-05,
+      "loss": 0.0012,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 0.1273522675037384,
+      "learning_rate": 2e-05,
+      "loss": 0.003,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 2069634366832640.0,
+      "train_loss": 0.07204394578933716,
+      "train_runtime": 101.8827,
+      "train_samples_per_second": 3.926,
+      "train_steps_per_second": 0.982
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 2069634366832640.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..5e360bb3b61d0353e8cd8bfed413c472493b1357
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2cd1af5c7ca40ed6a79af8c9cb6555f78589b0a624a98c6483565f76f8b3c61d
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..51c281c1fc7c1826002ca2df1569c5dde67688a6
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:8e8db792016ac55ff57bd23f8349e5df6ceb14624e1d9a29105c81ed7d7ed4ed
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..6fcbec7952fb9d959be011fbffce119e05716c98
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:19524fc38015f33290f292a4f522b4d36c1adcb621ac578bb561050e13fefe91
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..160ec64cd29c0e78f47a5a45e4a4d8705b97ceec
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:3a874d7d599aad5817e884d0953977542823faf1f01d25101400d3430928ff9c
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..f7c98fca20e49322bd9e8ccc52ce37ade7171177
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:33baa7cb694bc7b9f9162330827005d81631575ea2e90cd715b6d42b50f475da
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..e7ccf8f9310363dc996fc1d520afc9a74b734929
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cb3be0c563957173b91a7ec86c8723fdb23acdadf722cf602bc3483c5d0b11ae
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3736050fe3f44af1498b2a433eca06bc26df8cfa
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:698b9afe0148feecdd937cdc3624b381dfe48c2a3a8083e2dfa29bfedd728c42
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..2136b8e130c9f55a31b07896326a44f6dbe91f14
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:633deab3c32fa4046bc1e4c0691c368ae244561291e78cbdba6062fdf7b77f16
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..7b89af726a08b198c89f1ad4113f29c05b66c7ac
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 1.765663981437683,
+      "learning_rate": 2e-05,
+      "loss": 0.1386,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 4.676456928253174,
+      "learning_rate": 2e-05,
+      "loss": 1.1946,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 2.184738874435425,
+      "learning_rate": 2e-05,
+      "loss": 0.9233,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 4.627042770385742,
+      "learning_rate": 2e-05,
+      "loss": 1.1015,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 5.915734767913818,
+      "learning_rate": 2e-05,
+      "loss": 0.663,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 6.462231636047363,
+      "learning_rate": 2e-05,
+      "loss": 1.0081,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 0.35552021861076355,
+      "learning_rate": 2e-05,
+      "loss": 0.3945,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 5.338579177856445,
+      "learning_rate": 2e-05,
+      "loss": 0.6744,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 1.783922791481018,
+      "learning_rate": 2e-05,
+      "loss": 0.4461,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 3.711841106414795,
+      "learning_rate": 2e-05,
+      "loss": 1.4118,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 0.9069531559944153,
+      "learning_rate": 2e-05,
+      "loss": 0.176,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 7.7085490226745605,
+      "learning_rate": 2e-05,
+      "loss": 0.5315,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 10.80366039276123,
+      "learning_rate": 2e-05,
+      "loss": 1.0346,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 6.07086181640625,
+      "learning_rate": 2e-05,
+      "loss": 0.3799,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 1.8229091167449951,
+      "learning_rate": 2e-05,
+      "loss": 0.1932,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 5.323653697967529,
+      "learning_rate": 2e-05,
+      "loss": 0.3793,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 4.461909770965576,
+      "learning_rate": 2e-05,
+      "loss": 0.6431,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 1.6867424249649048,
+      "learning_rate": 2e-05,
+      "loss": 0.1087,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 3.3799192905426025,
+      "learning_rate": 2e-05,
+      "loss": 0.6874,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 7.544458389282227,
+      "learning_rate": 2e-05,
+      "loss": 0.6817,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 2.379153251647949,
+      "learning_rate": 2e-05,
+      "loss": 0.6136,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 2.67325758934021,
+      "learning_rate": 2e-05,
+      "loss": 0.1987,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 1.6791194677352905,
+      "learning_rate": 2e-05,
+      "loss": 0.3694,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 0.7336940169334412,
+      "learning_rate": 2e-05,
+      "loss": 0.3386,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 2.389741897583008,
+      "learning_rate": 2e-05,
+      "loss": 0.1778,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 7.384143829345703,
+      "learning_rate": 2e-05,
+      "loss": 0.7985,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 7.0400872230529785,
+      "learning_rate": 2e-05,
+      "loss": 0.6097,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 8.808256149291992,
+      "learning_rate": 2e-05,
+      "loss": 0.7863,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 3.6529507637023926,
+      "learning_rate": 2e-05,
+      "loss": 0.1368,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 0.17764848470687866,
+      "learning_rate": 2e-05,
+      "loss": 0.1653,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 1.8120052814483643,
+      "learning_rate": 2e-05,
+      "loss": 0.7154,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 3.756453037261963,
+      "learning_rate": 2e-05,
+      "loss": 0.5272,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 3.120513677597046,
+      "learning_rate": 2e-05,
+      "loss": 0.1667,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 4.981636047363281,
+      "learning_rate": 2e-05,
+      "loss": 0.3927,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 5.101208209991455,
+      "learning_rate": 2e-05,
+      "loss": 0.6766,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 1.0963003635406494,
+      "learning_rate": 2e-05,
+      "loss": 0.7616,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 0.7717503905296326,
+      "learning_rate": 2e-05,
+      "loss": 0.0356,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 0.8139798641204834,
+      "learning_rate": 2e-05,
+      "loss": 0.1705,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 3.033977746963501,
+      "learning_rate": 2e-05,
+      "loss": 0.1349,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 13.496634483337402,
+      "learning_rate": 2e-05,
+      "loss": 1.6624,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 13.66792106628418,
+      "learning_rate": 2e-05,
+      "loss": 2.9498,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 0.06267403066158295,
+      "learning_rate": 2e-05,
+      "loss": 0.0111,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 5.407293319702148,
+      "learning_rate": 2e-05,
+      "loss": 0.5322,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 6.728641510009766,
+      "learning_rate": 2e-05,
+      "loss": 0.6077,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 5.393722057342529,
+      "learning_rate": 2e-05,
+      "loss": 1.6041,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 0.15307481586933136,
+      "learning_rate": 2e-05,
+      "loss": 0.3262,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 6.039478302001953,
+      "learning_rate": 2e-05,
+      "loss": 0.491,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 3.6059703826904297,
+      "learning_rate": 2e-05,
+      "loss": 0.2965,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 10.516265869140625,
+      "learning_rate": 2e-05,
+      "loss": 1.887,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 2.030522584915161,
+      "learning_rate": 2e-05,
+      "loss": 0.1721,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 4914533793529856.0,
+      "train_loss": 0.6217503023147583,
+      "train_runtime": 169.8273,
+      "train_samples_per_second": 2.355,
+      "train_steps_per_second": 0.589
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 4914533793529856.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..b86edd5c0869c5a69a12bf0c6adeeda4c3b1afc2
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:386840b3ff94a28b7c83212829e6accfe75f2e328a9057cab81f329994fef0f1
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..4b425dbe03d5d986d9d1b898db1a609e55ba971e
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:99fe6a5d31215c1d0c764a74fdc4ec5918e52a0db960e55f5fdd3fe51f8dc2d2
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..22fe3ca766ec6e37d51aebf525dec2af34358ac0
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:924e12ae251af8a13eb6310668a2e9dbdc07b477f9b1ade3564b97b41171fb39
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3199fddd167e3c1256b44534489929ba61caa48f
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:dc65366e8dea534e149ce7cd583b6319994ed35ecc39582525344b753aeb1858
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..fa78c77c606806acc9895b6df49678c13633002e
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:92099a5ace36a30279e1cc431967135db0772dc3159d7b8d231675dfd4cf88c0
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..d9f75051c9cb79168d4847860a7429370a81c511
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:d874d9cb32ec1bddc888cfd18e4b414cf20eff538ba46baccd8b8d0b89bc4fbd
+size 184221358
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..98c30135b512ee9fd0e185bb6bdfee549aa02cae
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:85cba7e8c401b0d413e0b2a269065ccf0c313caf14e5148232105b7da2d2fa8f
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a7a1fbf8770ff4d959baf415401999d90d775e16
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b6a290cb3bf0eb2e5195b515d1ca2991a50522ba0f05ad0956cc52b0f6922957
+size 184220842
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..e84604826f0e299fa6c978a9f5b3f2eac8699cc0
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 11.693649291992188,
+      "learning_rate": 2e-05,
+      "loss": 1.0507,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 2.9311091899871826,
+      "learning_rate": 2e-05,
+      "loss": 0.4164,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 8.45428466796875,
+      "learning_rate": 2e-05,
+      "loss": 1.1758,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 14.28290843963623,
+      "learning_rate": 2e-05,
+      "loss": 1.3931,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 15.484107971191406,
+      "learning_rate": 2e-05,
+      "loss": 1.6668,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 9.924738883972168,
+      "learning_rate": 2e-05,
+      "loss": 1.5835,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 8.884476661682129,
+      "learning_rate": 2e-05,
+      "loss": 1.1145,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 9.459785461425781,
+      "learning_rate": 2e-05,
+      "loss": 0.9232,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 9.546226501464844,
+      "learning_rate": 2e-05,
+      "loss": 0.5492,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 2.949314594268799,
+      "learning_rate": 2e-05,
+      "loss": 0.744,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 5.894227027893066,
+      "learning_rate": 2e-05,
+      "loss": 0.6446,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 5.924652099609375,
+      "learning_rate": 2e-05,
+      "loss": 0.8909,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 2.6994128227233887,
+      "learning_rate": 2e-05,
+      "loss": 0.4439,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 8.177955627441406,
+      "learning_rate": 2e-05,
+      "loss": 0.4963,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 9.719780921936035,
+      "learning_rate": 2e-05,
+      "loss": 0.8525,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 22.681800842285156,
+      "learning_rate": 2e-05,
+      "loss": 1.8416,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 3.0644638538360596,
+      "learning_rate": 2e-05,
+      "loss": 2.1517,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 14.563640594482422,
+      "learning_rate": 2e-05,
+      "loss": 2.4433,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 7.03754186630249,
+      "learning_rate": 2e-05,
+      "loss": 1.3491,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 6.36046838760376,
+      "learning_rate": 2e-05,
+      "loss": 1.1212,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 18.849185943603516,
+      "learning_rate": 2e-05,
+      "loss": 1.6678,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 14.333285331726074,
+      "learning_rate": 2e-05,
+      "loss": 1.6081,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 6.0595269203186035,
+      "learning_rate": 2e-05,
+      "loss": 0.6205,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 4.74971342086792,
+      "learning_rate": 2e-05,
+      "loss": 0.4097,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 7.75990104675293,
+      "learning_rate": 2e-05,
+      "loss": 1.074,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 9.010882377624512,
+      "learning_rate": 2e-05,
+      "loss": 2.1047,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 1.1840580701828003,
+      "learning_rate": 2e-05,
+      "loss": 0.8681,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 2.5251049995422363,
+      "learning_rate": 2e-05,
+      "loss": 0.7495,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 6.284056186676025,
+      "learning_rate": 2e-05,
+      "loss": 0.8752,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 8.168758392333984,
+      "learning_rate": 2e-05,
+      "loss": 1.3925,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 6.6547770500183105,
+      "learning_rate": 2e-05,
+      "loss": 0.7414,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 7.461399078369141,
+      "learning_rate": 2e-05,
+      "loss": 0.8258,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 6.963188648223877,
+      "learning_rate": 2e-05,
+      "loss": 1.218,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 5.871964454650879,
+      "learning_rate": 2e-05,
+      "loss": 1.1124,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 7.110522747039795,
+      "learning_rate": 2e-05,
+      "loss": 1.2579,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 8.83693790435791,
+      "learning_rate": 2e-05,
+      "loss": 2.2598,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 7.1040520668029785,
+      "learning_rate": 2e-05,
+      "loss": 1.2506,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 17.06218719482422,
+      "learning_rate": 2e-05,
+      "loss": 1.4432,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 7.473016262054443,
+      "learning_rate": 2e-05,
+      "loss": 0.784,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 8.905854225158691,
+      "learning_rate": 2e-05,
+      "loss": 0.8411,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 4.500631332397461,
+      "learning_rate": 2e-05,
+      "loss": 0.2397,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 4.327226638793945,
+      "learning_rate": 2e-05,
+      "loss": 0.4464,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 4.575321674346924,
+      "learning_rate": 2e-05,
+      "loss": 1.1319,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 4.218846321105957,
+      "learning_rate": 2e-05,
+      "loss": 0.8927,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 7.213256359100342,
+      "learning_rate": 2e-05,
+      "loss": 0.9017,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 10.291433334350586,
+      "learning_rate": 2e-05,
+      "loss": 0.6296,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 6.705934524536133,
+      "learning_rate": 2e-05,
+      "loss": 0.6399,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 3.8862531185150146,
+      "learning_rate": 2e-05,
+      "loss": 0.3221,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 9.070793151855469,
+      "learning_rate": 2e-05,
+      "loss": 1.1441,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 11.29675006866455,
+      "learning_rate": 2e-05,
+      "loss": 1.2231,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 2097655350034432.0,
+      "train_loss": 1.0705613040924071,
+      "train_runtime": 102.5888,
+      "train_samples_per_second": 3.899,
+      "train_steps_per_second": 0.975
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 2097655350034432.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..ea26972b174f00ea7b91a4336211d00f1bbf6ccd
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:63c020f0e9a8f0c812148e6b98c897be1fbf16e0e6bdf9c1cb855bd2fb184967
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..85736a412801b337f711c8a1f39a4881e5956574
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:a576f6efc9cbd3bb8a501153bc3e085ec89a15902ad8cfe10bc4b974f03e4fcd
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..9f0a7334bc0ea07ab1adafbe7282923c55639e0d
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5db05f3d3601799e1fc60bea63f67d5557eee27fabad09326273c4391caa1f03
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a1fc7bc7abc59c62ff76a6fa1523d070b0127f9e
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7b1615aeafa8813c4d2376f2ae7a14b957139e2e63d429b39c3fb322e8bde63a
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..b4660c6430d4841f8c2e176a79ac3cdd3e7b7cd9
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6ba947d6751d0af4d240a2ac8b103cec9438b64bef846a3c69fd34bf8554f5df
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..275ba55dd79e0c9eb47597a29f29c41b2c2f3084
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:56fa94a36ca7fe11a494c594eb9b640dd6ee9b47daeee27e2fee4d62575c6f43
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..5a64572f2ea255406eea5affb513586158bc3f83
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ac555e618f6ec228443d7e19fcd469b8ef8e5bea20f651ef5cde9cdf85221ad9
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..c818a8f3a31058b6182a391af6a4a66ac293729d
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9e7a27691ec0a074141418aadc3a9c16e25bbeb6a18c82bc76883e6acb92cb24
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..cce23ee7dc473a1a85953a45ba7c9ab5805ada89
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 2.637403726577759,
+      "learning_rate": 2e-05,
+      "loss": 0.5614,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 4.73873233795166,
+      "learning_rate": 2e-05,
+      "loss": 0.9697,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 1.6665256023406982,
+      "learning_rate": 2e-05,
+      "loss": 0.4683,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 3.0230000019073486,
+      "learning_rate": 2e-05,
+      "loss": 0.6475,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 2.887681245803833,
+      "learning_rate": 2e-05,
+      "loss": 0.5517,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 3.429380416870117,
+      "learning_rate": 2e-05,
+      "loss": 0.2907,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 4.610021114349365,
+      "learning_rate": 2e-05,
+      "loss": 0.6544,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 8.516341209411621,
+      "learning_rate": 2e-05,
+      "loss": 1.9958,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 3.5461740493774414,
+      "learning_rate": 2e-05,
+      "loss": 0.3917,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 5.2099456787109375,
+      "learning_rate": 2e-05,
+      "loss": 0.4288,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 4.585675239562988,
+      "learning_rate": 2e-05,
+      "loss": 1.801,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 2.1597061157226562,
+      "learning_rate": 2e-05,
+      "loss": 1.0035,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 3.1741673946380615,
+      "learning_rate": 2e-05,
+      "loss": 0.4851,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 5.476759433746338,
+      "learning_rate": 2e-05,
+      "loss": 1.1165,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 4.2731804847717285,
+      "learning_rate": 2e-05,
+      "loss": 0.9401,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 6.725061893463135,
+      "learning_rate": 2e-05,
+      "loss": 0.7967,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 2.4326884746551514,
+      "learning_rate": 2e-05,
+      "loss": 0.4525,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 4.693164348602295,
+      "learning_rate": 2e-05,
+      "loss": 1.2137,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 2.8988044261932373,
+      "learning_rate": 2e-05,
+      "loss": 0.5093,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 6.970594882965088,
+      "learning_rate": 2e-05,
+      "loss": 1.0991,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 3.4238510131835938,
+      "learning_rate": 2e-05,
+      "loss": 0.8502,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 3.9824445247650146,
+      "learning_rate": 2e-05,
+      "loss": 0.6589,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 4.8126678466796875,
+      "learning_rate": 2e-05,
+      "loss": 0.8948,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 5.100122451782227,
+      "learning_rate": 2e-05,
+      "loss": 0.8117,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 2.1359190940856934,
+      "learning_rate": 2e-05,
+      "loss": 0.4148,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 5.385488033294678,
+      "learning_rate": 2e-05,
+      "loss": 1.6419,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 4.634507656097412,
+      "learning_rate": 2e-05,
+      "loss": 0.8469,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 2.4589309692382812,
+      "learning_rate": 2e-05,
+      "loss": 0.2448,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 4.612209796905518,
+      "learning_rate": 2e-05,
+      "loss": 0.836,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 3.2502317428588867,
+      "learning_rate": 2e-05,
+      "loss": 0.8215,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 3.3926491737365723,
+      "learning_rate": 2e-05,
+      "loss": 0.6394,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 3.6580755710601807,
+      "learning_rate": 2e-05,
+      "loss": 1.1433,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 7.191780090332031,
+      "learning_rate": 2e-05,
+      "loss": 1.0154,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 2.0875039100646973,
+      "learning_rate": 2e-05,
+      "loss": 0.6591,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 7.09459114074707,
+      "learning_rate": 2e-05,
+      "loss": 0.8218,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 3.325845241546631,
+      "learning_rate": 2e-05,
+      "loss": 0.6073,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 1.3921314477920532,
+      "learning_rate": 2e-05,
+      "loss": 0.5865,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 3.6516613960266113,
+      "learning_rate": 2e-05,
+      "loss": 1.4063,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 4.696588039398193,
+      "learning_rate": 2e-05,
+      "loss": 0.7179,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 3.545356512069702,
+      "learning_rate": 2e-05,
+      "loss": 0.4085,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 3.8088479042053223,
+      "learning_rate": 2e-05,
+      "loss": 0.5266,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 5.2242326736450195,
+      "learning_rate": 2e-05,
+      "loss": 0.6016,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 2.499507427215576,
+      "learning_rate": 2e-05,
+      "loss": 0.3916,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 3.5232198238372803,
+      "learning_rate": 2e-05,
+      "loss": 0.5096,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 4.562027931213379,
+      "learning_rate": 2e-05,
+      "loss": 0.7677,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 6.415626525878906,
+      "learning_rate": 2e-05,
+      "loss": 0.7187,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 5.6967644691467285,
+      "learning_rate": 2e-05,
+      "loss": 0.7985,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 3.2610225677490234,
+      "learning_rate": 2e-05,
+      "loss": 0.6187,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 3.403749942779541,
+      "learning_rate": 2e-05,
+      "loss": 0.2896,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 11.633613586425781,
+      "learning_rate": 2e-05,
+      "loss": 1.1527,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 5694661670731776.0,
+      "train_loss": 0.7755919742584229,
+      "train_runtime": 170.0795,
+      "train_samples_per_second": 2.352,
+      "train_steps_per_second": 0.588
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 5694661670731776.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..875ef31ff0602658917c89cbfc58d6319d4d4d37
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:75fce352d5052d2b03312c457b9d07a3c2045a3c1d82fa169cde815ffac140a0
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..c5932b68864f01c99fd5f5b82e6e4cb611e4240a
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:51efd3d761dbc6a85da1d36fd13afc6913e8f35c7c18b9825c9a03ebd8725119
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..b08ea826b7c2c88fd2b5c85ee58a048c4c3540b1
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:04acc4730d61aa5cab2b5961a914144dea9077c80a98ea800ef7b634fb202740
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..efb36b99cc3c7422754745d8810c9c83a382d36d
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c21f2c0bf74b5e28d521eff1cae89902767d9acad0a560a77818c53885c72763
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..599b0e6863625123e3cdd051bbadd221db394fea
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:cf27c9f6c67ee9af8765ac0ea5533f11f5d1e729b6e3d53daf4b640cb3853670
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..d2ed6d0b38929d9f0b693a83ff3b6c6d651b0bca
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:f83aa707ae161bc1ddd50bd1a996c1f78a7d97983e4fdb7e4ba8894b96b33eda
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..7632e8315ae00320363c5e98e76244b18bc9cce9
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:04d4438806ea71f5e33bb20506679a54f6c0badd58b2e989c7ff596a42caa40b
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..b53da66277ada30459c7744639680cd00357a4bc
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:7d0ab48aaaf037776c5d89c9130d080f2274777ef2ae7806989abda2477652e0
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c2ad527d1c30d44d2dbb2f8bd1981fae03e072da
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 1.0338062047958374,
+      "learning_rate": 2e-05,
+      "loss": 0.7566,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 5.665828704833984,
+      "learning_rate": 2e-05,
+      "loss": 1.0194,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 6.612151145935059,
+      "learning_rate": 2e-05,
+      "loss": 1.6411,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 3.471060276031494,
+      "learning_rate": 2e-05,
+      "loss": 0.5854,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 1.2747403383255005,
+      "learning_rate": 2e-05,
+      "loss": 0.4297,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 8.470596313476562,
+      "learning_rate": 2e-05,
+      "loss": 1.0594,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 0.36429786682128906,
+      "learning_rate": 2e-05,
+      "loss": 0.3329,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 4.337061882019043,
+      "learning_rate": 2e-05,
+      "loss": 0.5275,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 5.764715671539307,
+      "learning_rate": 2e-05,
+      "loss": 1.6987,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 3.9755842685699463,
+      "learning_rate": 2e-05,
+      "loss": 1.864,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 7.792990207672119,
+      "learning_rate": 2e-05,
+      "loss": 3.8483,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 6.800496578216553,
+      "learning_rate": 2e-05,
+      "loss": 1.0842,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 5.43037223815918,
+      "learning_rate": 2e-05,
+      "loss": 1.2437,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 6.251035690307617,
+      "learning_rate": 2e-05,
+      "loss": 2.8937,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 0.5180622339248657,
+      "learning_rate": 2e-05,
+      "loss": 0.3756,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 1.7391629219055176,
+      "learning_rate": 2e-05,
+      "loss": 0.4611,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 1.569304347038269,
+      "learning_rate": 2e-05,
+      "loss": 0.215,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 6.430338382720947,
+      "learning_rate": 2e-05,
+      "loss": 1.2636,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 4.805810928344727,
+      "learning_rate": 2e-05,
+      "loss": 1.105,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 4.90680456161499,
+      "learning_rate": 2e-05,
+      "loss": 0.5404,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 3.3493094444274902,
+      "learning_rate": 2e-05,
+      "loss": 0.6253,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 5.143937110900879,
+      "learning_rate": 2e-05,
+      "loss": 0.7676,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 2.285219669342041,
+      "learning_rate": 2e-05,
+      "loss": 0.7972,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 7.806530475616455,
+      "learning_rate": 2e-05,
+      "loss": 3.4581,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 12.248115539550781,
+      "learning_rate": 2e-05,
+      "loss": 4.6998,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 5.186307907104492,
+      "learning_rate": 2e-05,
+      "loss": 2.4965,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 3.7022056579589844,
+      "learning_rate": 2e-05,
+      "loss": 0.5178,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 13.997264862060547,
+      "learning_rate": 2e-05,
+      "loss": 1.0521,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 7.620451927185059,
+      "learning_rate": 2e-05,
+      "loss": 2.9059,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 2.462949514389038,
+      "learning_rate": 2e-05,
+      "loss": 0.2984,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 3.5282416343688965,
+      "learning_rate": 2e-05,
+      "loss": 1.3804,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 7.946673393249512,
+      "learning_rate": 2e-05,
+      "loss": 1.8044,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 3.367640495300293,
+      "learning_rate": 2e-05,
+      "loss": 0.6136,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 3.411672592163086,
+      "learning_rate": 2e-05,
+      "loss": 0.8734,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 6.9260969161987305,
+      "learning_rate": 2e-05,
+      "loss": 3.405,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 2.8345489501953125,
+      "learning_rate": 2e-05,
+      "loss": 2.5527,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 0.7165583372116089,
+      "learning_rate": 2e-05,
+      "loss": 0.4615,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 8.22752857208252,
+      "learning_rate": 2e-05,
+      "loss": 0.9082,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 4.878330707550049,
+      "learning_rate": 2e-05,
+      "loss": 1.7125,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 6.852156639099121,
+      "learning_rate": 2e-05,
+      "loss": 1.0581,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 4.600001335144043,
+      "learning_rate": 2e-05,
+      "loss": 0.8398,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 4.03223991394043,
+      "learning_rate": 2e-05,
+      "loss": 0.7341,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 2.608586072921753,
+      "learning_rate": 2e-05,
+      "loss": 0.3577,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 7.7058424949646,
+      "learning_rate": 2e-05,
+      "loss": 0.9836,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 2.9475879669189453,
+      "learning_rate": 2e-05,
+      "loss": 0.3618,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 7.979467391967773,
+      "learning_rate": 2e-05,
+      "loss": 1.1729,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 5.021623611450195,
+      "learning_rate": 2e-05,
+      "loss": 0.5304,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 10.417924880981445,
+      "learning_rate": 2e-05,
+      "loss": 5.0102,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 0.491908997297287,
+      "learning_rate": 2e-05,
+      "loss": 2.575,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 7.822747707366943,
+      "learning_rate": 2e-05,
+      "loss": 1.7126,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 4901658932805632.0,
+      "train_loss": 1.3922410774230958,
+      "train_runtime": 170.2011,
+      "train_samples_per_second": 2.35,
+      "train_steps_per_second": 0.588
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 4901658932805632.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..450048d95703dc653eb333567ab2f9cd68d07eb6
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:96f6f7356fb9a5a9dd2a32d459145266c4872f8dafddd2a1b42c364247ec8d1c
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a6fbe39e373bdbfd107dae3782b7893703b21626
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:25c6be73f4fc66311d77d70650419c2a97c8790f3c3d84f647af59dc02768e4e
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3397f65713a1de89bca9d576095b085df121e608
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:c81e7e0ad944248aa4f4b98e41afbd836df459feca97c8fb9707bd6b80b71ce1
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..89ff0dd61545afe45726fbee8c2c99b3427f4949
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ec0a8f4ff9daea9e7beb8c98c1b0ec4d9095c883738a1088014aaa0a7bf48f8a
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..12160a9a529958da92318811e514ce6b95c06abe
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:98715cd26748066462045c4642894e1027030461bb77ad87770cc76280c54a3b
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..05982cc8fe8038f2f156890af8c33a408ad19a64
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:b58f5e3640f9bff8535d561cccaedb6f3ae0d6ab6e7ec2f903b0fe5476b7d5a2
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..fe85a8db8d0505b352d5733159222903fa505b85
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:853c02dd7bd55df85125d54791c6918907a6fdf90f4ad1da31f424ec046d79b8
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..2658e5557745e538a528e488844cab647ce89913
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:6b1a1d15929d7074e34afe26791a162fde4f54760b447fa045085f2bb515bfce
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..c5e5671291f6d8348dc478dff6c19229dd636c87
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/6_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 6.350866794586182,
+      "learning_rate": 2e-05,
+      "loss": 0.8442,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 7.095775604248047,
+      "learning_rate": 2e-05,
+      "loss": 0.862,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 6.973437786102295,
+      "learning_rate": 2e-05,
+      "loss": 1.0801,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 2.7381036281585693,
+      "learning_rate": 2e-05,
+      "loss": 0.8341,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 5.858903408050537,
+      "learning_rate": 2e-05,
+      "loss": 1.3088,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 1.146891474723816,
+      "learning_rate": 2e-05,
+      "loss": 0.1271,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 5.8700642585754395,
+      "learning_rate": 2e-05,
+      "loss": 0.6333,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 10.154126167297363,
+      "learning_rate": 2e-05,
+      "loss": 1.3238,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 7.10991096496582,
+      "learning_rate": 2e-05,
+      "loss": 1.7668,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 4.042394161224365,
+      "learning_rate": 2e-05,
+      "loss": 0.7352,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 4.039437770843506,
+      "learning_rate": 2e-05,
+      "loss": 0.792,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 1.5868761539459229,
+      "learning_rate": 2e-05,
+      "loss": 0.7132,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 2.3178021907806396,
+      "learning_rate": 2e-05,
+      "loss": 0.7587,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 6.503546237945557,
+      "learning_rate": 2e-05,
+      "loss": 0.8239,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 6.615965366363525,
+      "learning_rate": 2e-05,
+      "loss": 1.5844,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 6.158043384552002,
+      "learning_rate": 2e-05,
+      "loss": 0.7461,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 3.276494026184082,
+      "learning_rate": 2e-05,
+      "loss": 0.9598,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 3.5142297744750977,
+      "learning_rate": 2e-05,
+      "loss": 1.0434,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 2.0729358196258545,
+      "learning_rate": 2e-05,
+      "loss": 0.8575,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 3.127547264099121,
+      "learning_rate": 2e-05,
+      "loss": 0.8279,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 3.0067636966705322,
+      "learning_rate": 2e-05,
+      "loss": 1.3029,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 4.825779914855957,
+      "learning_rate": 2e-05,
+      "loss": 0.5708,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 5.970290184020996,
+      "learning_rate": 2e-05,
+      "loss": 0.7744,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 5.233969211578369,
+      "learning_rate": 2e-05,
+      "loss": 0.7049,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 3.52846622467041,
+      "learning_rate": 2e-05,
+      "loss": 0.9376,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 1.7202891111373901,
+      "learning_rate": 2e-05,
+      "loss": 0.749,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 4.498488903045654,
+      "learning_rate": 2e-05,
+      "loss": 0.7168,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 5.07948112487793,
+      "learning_rate": 2e-05,
+      "loss": 1.7681,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 7.285374164581299,
+      "learning_rate": 2e-05,
+      "loss": 0.9782,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 3.47748064994812,
+      "learning_rate": 2e-05,
+      "loss": 1.2967,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 5.545077800750732,
+      "learning_rate": 2e-05,
+      "loss": 0.9142,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 2.9953062534332275,
+      "learning_rate": 2e-05,
+      "loss": 0.8001,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 2.4720499515533447,
+      "learning_rate": 2e-05,
+      "loss": 1.0479,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 6.32978630065918,
+      "learning_rate": 2e-05,
+      "loss": 1.0165,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 3.1678273677825928,
+      "learning_rate": 2e-05,
+      "loss": 1.0068,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 1.8918789625167847,
+      "learning_rate": 2e-05,
+      "loss": 0.6592,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 7.592050075531006,
+      "learning_rate": 2e-05,
+      "loss": 1.2711,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 4.536243915557861,
+      "learning_rate": 2e-05,
+      "loss": 1.1003,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 2.779127836227417,
+      "learning_rate": 2e-05,
+      "loss": 0.494,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 3.7273905277252197,
+      "learning_rate": 2e-05,
+      "loss": 0.4256,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 3.0089080333709717,
+      "learning_rate": 2e-05,
+      "loss": 0.8304,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 3.3422775268554688,
+      "learning_rate": 2e-05,
+      "loss": 1.3364,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 4.476714134216309,
+      "learning_rate": 2e-05,
+      "loss": 0.4442,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 2.4386043548583984,
+      "learning_rate": 2e-05,
+      "loss": 0.8881,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 6.396787166595459,
+      "learning_rate": 2e-05,
+      "loss": 0.5421,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 2.9071009159088135,
+      "learning_rate": 2e-05,
+      "loss": 2.2692,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 2.4817564487457275,
+      "learning_rate": 2e-05,
+      "loss": 0.5462,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 10.91292953491211,
+      "learning_rate": 2e-05,
+      "loss": 1.1156,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 7.508020401000977,
+      "learning_rate": 2e-05,
+      "loss": 1.2463,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 7.0632500648498535,
+      "learning_rate": 2e-05,
+      "loss": 1.967,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 9847400269086720.0,
+      "train_loss": 0.9668588447570801,
+      "train_runtime": 198.3477,
+      "train_samples_per_second": 2.017,
+      "train_steps_per_second": 0.504
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 9847400269086720.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..ecd3324a272ad2e19c7c5beb80029382e5e9281a
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:78b485308edf5a924f2ef57a05c65017aa735f4f8f751d6d24147be61a379cdd
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..d3cef4e2b408a96fa76cf196961b568845bea1db
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:13a51c7a63f8ddcbe3f0878625d00b95735ba43292babc2a0968ae38b3222b8f
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..4402ece2c150a460dfb18019ce1b7e10c7e2155d
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:15f95e115c2ab8cc0417b6de7b16011a72e457dd83bff996f6e0595d94aca276
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3089429fa4b6f8ec7c27c19c37481eaabb9c575f
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:ce54c06b61b3bb1eb86188b4a03197e6d959d69b5cc7b6e72d2969d352596b74
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..48f2225e8df0876ec24175e4680df9ce285f7e69
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:74adb04a0f7c2f9bd64a9979982262c86e0a658cdb2a126b4d1af7db859602f0
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..ac7154657cde3cbfe9ea0745960d39c5fa2bbd64
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:27a43238a35228ee66d44ddf92b40183b15124a43d59e49640cb47ed81a86b68
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..a70fd99732154db654a6694c9026e74e127e7a48
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:9b078ac3ebb4966747b773b17e3a0b4a019db00d3e099deba6c28f83ce3a7257
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..9d1d167d0bf4481186a8c85c09d03169d8f0034a
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:0b9e95bcd67b3d6f50550885ae19aac23a3a1a33c664205e2ac601aa0367f43c
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..b493ce4acbee516d93764e564063ce12eead09e4
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/7_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 2.5717689990997314,
+      "learning_rate": 2e-05,
+      "loss": 0.1378,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 6.968014717102051,
+      "learning_rate": 2e-05,
+      "loss": 0.7146,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 4.446895599365234,
+      "learning_rate": 2e-05,
+      "loss": 0.3686,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 1.4856871366500854,
+      "learning_rate": 2e-05,
+      "loss": 0.5728,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 4.265855312347412,
+      "learning_rate": 2e-05,
+      "loss": 1.149,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 0.34307774901390076,
+      "learning_rate": 2e-05,
+      "loss": 0.1337,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 0.6195315718650818,
+      "learning_rate": 2e-05,
+      "loss": 0.1026,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 3.3206229209899902,
+      "learning_rate": 2e-05,
+      "loss": 0.8001,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 4.18081521987915,
+      "learning_rate": 2e-05,
+      "loss": 0.3459,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 6.578738689422607,
+      "learning_rate": 2e-05,
+      "loss": 0.4957,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 1.6412115097045898,
+      "learning_rate": 2e-05,
+      "loss": 0.3151,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 5.7735209465026855,
+      "learning_rate": 2e-05,
+      "loss": 0.4686,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 3.0710041522979736,
+      "learning_rate": 2e-05,
+      "loss": 0.1932,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 2.2720937728881836,
+      "learning_rate": 2e-05,
+      "loss": 0.3962,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 0.873317301273346,
+      "learning_rate": 2e-05,
+      "loss": 1.1774,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 2.1178581714630127,
+      "learning_rate": 2e-05,
+      "loss": 0.5692,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 4.69739294052124,
+      "learning_rate": 2e-05,
+      "loss": 0.2105,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 5.956394195556641,
+      "learning_rate": 2e-05,
+      "loss": 1.0291,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 0.9587688446044922,
+      "learning_rate": 2e-05,
+      "loss": 0.1533,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 9.836400985717773,
+      "learning_rate": 2e-05,
+      "loss": 1.6179,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 0.8551999926567078,
+      "learning_rate": 2e-05,
+      "loss": 0.7573,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 4.6954827308654785,
+      "learning_rate": 2e-05,
+      "loss": 0.2751,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 3.2609949111938477,
+      "learning_rate": 2e-05,
+      "loss": 0.699,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 0.6860560774803162,
+      "learning_rate": 2e-05,
+      "loss": 0.0301,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 0.28854402899742126,
+      "learning_rate": 2e-05,
+      "loss": 1.0212,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 4.731658935546875,
+      "learning_rate": 2e-05,
+      "loss": 1.2925,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 0.11632199585437775,
+      "learning_rate": 2e-05,
+      "loss": 0.8402,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 0.5225258469581604,
+      "learning_rate": 2e-05,
+      "loss": 0.047,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 0.5997475981712341,
+      "learning_rate": 2e-05,
+      "loss": 0.3446,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 10.83228588104248,
+      "learning_rate": 2e-05,
+      "loss": 2.1025,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 0.7379941940307617,
+      "learning_rate": 2e-05,
+      "loss": 0.0965,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 5.670101642608643,
+      "learning_rate": 2e-05,
+      "loss": 0.9854,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 1.83771550655365,
+      "learning_rate": 2e-05,
+      "loss": 0.2185,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 2.517608642578125,
+      "learning_rate": 2e-05,
+      "loss": 0.2315,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 6.244460582733154,
+      "learning_rate": 2e-05,
+      "loss": 0.9419,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 4.1970391273498535,
+      "learning_rate": 2e-05,
+      "loss": 0.6381,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 7.284065246582031,
+      "learning_rate": 2e-05,
+      "loss": 1.0512,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 1.3797696828842163,
+      "learning_rate": 2e-05,
+      "loss": 0.7621,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 1.6824308633804321,
+      "learning_rate": 2e-05,
+      "loss": 0.7541,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 8.63947582244873,
+      "learning_rate": 2e-05,
+      "loss": 1.9929,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 2.2472915649414062,
+      "learning_rate": 2e-05,
+      "loss": 0.2506,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 3.2342183589935303,
+      "learning_rate": 2e-05,
+      "loss": 0.3462,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 3.877340078353882,
+      "learning_rate": 2e-05,
+      "loss": 1.3992,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 7.802109718322754,
+      "learning_rate": 2e-05,
+      "loss": 1.2399,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 3.077086925506592,
+      "learning_rate": 2e-05,
+      "loss": 1.1619,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 0.8278634548187256,
+      "learning_rate": 2e-05,
+      "loss": 0.3156,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 1.945942997932434,
+      "learning_rate": 2e-05,
+      "loss": 0.1714,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 0.2935602366924286,
+      "learning_rate": 2e-05,
+      "loss": 0.2344,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 2.8402061462402344,
+      "learning_rate": 2e-05,
+      "loss": 0.3634,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 5.449839115142822,
+      "learning_rate": 2e-05,
+      "loss": 0.455,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 5158236269314048.0,
+      "train_loss": 0.6394131851196289,
+      "train_runtime": 170.8215,
+      "train_samples_per_second": 2.342,
+      "train_steps_per_second": 0.585
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 5158236269314048.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round10.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round10.pth
new file mode 100644
index 0000000000000000000000000000000000000000..e1561a8285a85251eaba0aa5e884a8b94d9ab767
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round10.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:5741134779b229b448c5e3fc8024d44bffdbd9ebca7fdcc3bd880aa835a31280
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round12.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round12.pth
new file mode 100644
index 0000000000000000000000000000000000000000..7e506cf512eeb98a8e3b0b437428c61677218065
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round12.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:2b374235edfcd56ee363e78bcea30bc6b0218b6c63c9bc3ddb516e6efcbe64eb
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round15.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round15.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3b8f5fc68d3bb62d79ecf1de1a1452f773780fe1
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round15.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:01f20d696391c4300f109f34470924e22885d200aed6b1c45cd2b11e3f60b61b
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round17.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round17.pth
new file mode 100644
index 0000000000000000000000000000000000000000..32227b60a637028df36cc93f527e712e55a021f7
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round17.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:47e49964d10a8d173547a0ce060575e8d0dc9a9b431f321f8aacd61a08e785a7
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round2.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round2.pth
new file mode 100644
index 0000000000000000000000000000000000000000..3215023ab2eb1a85d819238dfaaa69910010fd11
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round2.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:025f4c77a9baea1bebbef6e42fff1c398590fb24c6e55c197cb55b50e94dd682
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round20.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round20.pth
new file mode 100644
index 0000000000000000000000000000000000000000..36de5320d0fc9c1cda83892249b0488e0cc53256
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round20.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:1abbefa222ae9042b4bb67523aec2c06a4b0917fe3bb41469a8bce791ea0cc84
+size 395787774
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round5.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round5.pth
new file mode 100644
index 0000000000000000000000000000000000000000..49253fdce3d90d924c117c0b02fbe4fb3cbaa1f0
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round5.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:e4eb8227422e278d8dfabde57874b10b62347351bfd9185e17b97be3ea560c38
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round7.pth b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round7.pth
new file mode 100644
index 0000000000000000000000000000000000000000..1a31f364d5a0b57e83c028d01ada716299d30203
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_client_model_round7.pth
@@ -0,0 +1,3 @@
+version https://git-lfs.github.com/spec/v1
+oid sha256:79cc6353eb47ef09f9cdd8d06935408c221f92ffcea3c01c2bb67086e201bbde
+size 395786922
diff --git a/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_trainer_state.json b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_trainer_state.json
new file mode 100644
index 0000000000000000000000000000000000000000..aa8b73f3ee317fdfcfc055ba5d42addebc5912c8
--- /dev/null
+++ b/client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/8_trainer_state.json
@@ -0,0 +1,392 @@
+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 100,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.02,
+      "grad_norm": 0.16113965213298798,
+      "learning_rate": 2e-05,
+      "loss": 0.0915,
+      "step": 2
+    },
+    {
+      "epoch": 0.04,
+      "grad_norm": 0.07829011231660843,
+      "learning_rate": 2e-05,
+      "loss": 0.1069,
+      "step": 4
+    },
+    {
+      "epoch": 0.06,
+      "grad_norm": 0.37092337012290955,
+      "learning_rate": 2e-05,
+      "loss": 0.131,
+      "step": 6
+    },
+    {
+      "epoch": 0.08,
+      "grad_norm": 3.931626796722412,
+      "learning_rate": 2e-05,
+      "loss": 0.4551,
+      "step": 8
+    },
+    {
+      "epoch": 0.1,
+      "grad_norm": 0.05373501405119896,
+      "learning_rate": 2e-05,
+      "loss": 0.0031,
+      "step": 10
+    },
+    {
+      "epoch": 0.12,
+      "grad_norm": 2.1131927967071533,
+      "learning_rate": 2e-05,
+      "loss": 0.1581,
+      "step": 12
+    },
+    {
+      "epoch": 0.14,
+      "grad_norm": 0.07835656404495239,
+      "learning_rate": 2e-05,
+      "loss": 0.3279,
+      "step": 14
+    },
+    {
+      "epoch": 0.16,
+      "grad_norm": 8.693585395812988,
+      "learning_rate": 2e-05,
+      "loss": 0.5947,
+      "step": 16
+    },
+    {
+      "epoch": 0.18,
+      "grad_norm": 2.7874717712402344,
+      "learning_rate": 2e-05,
+      "loss": 0.2964,
+      "step": 18
+    },
+    {
+      "epoch": 0.2,
+      "grad_norm": 6.932012557983398,
+      "learning_rate": 2e-05,
+      "loss": 3.8206,
+      "step": 20
+    },
+    {
+      "epoch": 0.22,
+      "grad_norm": 5.361955642700195,
+      "learning_rate": 2e-05,
+      "loss": 0.7578,
+      "step": 22
+    },
+    {
+      "epoch": 0.24,
+      "grad_norm": 0.567628800868988,
+      "learning_rate": 2e-05,
+      "loss": 0.9104,
+      "step": 24
+    },
+    {
+      "epoch": 0.26,
+      "grad_norm": 2.3103795051574707,
+      "learning_rate": 2e-05,
+      "loss": 0.157,
+      "step": 26
+    },
+    {
+      "epoch": 0.28,
+      "grad_norm": 1.790549874305725,
+      "learning_rate": 2e-05,
+      "loss": 0.1,
+      "step": 28
+    },
+    {
+      "epoch": 0.3,
+      "grad_norm": 2.410795211791992,
+      "learning_rate": 2e-05,
+      "loss": 0.1141,
+      "step": 30
+    },
+    {
+      "epoch": 0.32,
+      "grad_norm": 0.02548924647271633,
+      "learning_rate": 2e-05,
+      "loss": 0.4958,
+      "step": 32
+    },
+    {
+      "epoch": 0.34,
+      "grad_norm": 4.374381065368652,
+      "learning_rate": 2e-05,
+      "loss": 1.8172,
+      "step": 34
+    },
+    {
+      "epoch": 0.36,
+      "grad_norm": 3.1204335689544678,
+      "learning_rate": 2e-05,
+      "loss": 0.6653,
+      "step": 36
+    },
+    {
+      "epoch": 0.38,
+      "grad_norm": 4.962135314941406,
+      "learning_rate": 2e-05,
+      "loss": 0.7358,
+      "step": 38
+    },
+    {
+      "epoch": 0.4,
+      "grad_norm": 2.9371933937072754,
+      "learning_rate": 2e-05,
+      "loss": 1.0185,
+      "step": 40
+    },
+    {
+      "epoch": 0.42,
+      "grad_norm": 6.699105739593506,
+      "learning_rate": 2e-05,
+      "loss": 0.4942,
+      "step": 42
+    },
+    {
+      "epoch": 0.44,
+      "grad_norm": 0.01869381032884121,
+      "learning_rate": 2e-05,
+      "loss": 0.0809,
+      "step": 44
+    },
+    {
+      "epoch": 0.46,
+      "grad_norm": 5.275974273681641,
+      "learning_rate": 2e-05,
+      "loss": 0.8099,
+      "step": 46
+    },
+    {
+      "epoch": 0.48,
+      "grad_norm": 6.025607109069824,
+      "learning_rate": 2e-05,
+      "loss": 1.2858,
+      "step": 48
+    },
+    {
+      "epoch": 0.5,
+      "grad_norm": 0.7212814688682556,
+      "learning_rate": 2e-05,
+      "loss": 0.4049,
+      "step": 50
+    },
+    {
+      "epoch": 0.52,
+      "grad_norm": 0.0265222005546093,
+      "learning_rate": 2e-05,
+      "loss": 0.4391,
+      "step": 52
+    },
+    {
+      "epoch": 0.54,
+      "grad_norm": 0.3079047203063965,
+      "learning_rate": 2e-05,
+      "loss": 0.7052,
+      "step": 54
+    },
+    {
+      "epoch": 0.56,
+      "grad_norm": 0.7660133838653564,
+      "learning_rate": 2e-05,
+      "loss": 0.4831,
+      "step": 56
+    },
+    {
+      "epoch": 0.58,
+      "grad_norm": 2.1217522621154785,
+      "learning_rate": 2e-05,
+      "loss": 0.1372,
+      "step": 58
+    },
+    {
+      "epoch": 0.6,
+      "grad_norm": 0.07789596170186996,
+      "learning_rate": 2e-05,
+      "loss": 0.012,
+      "step": 60
+    },
+    {
+      "epoch": 0.62,
+      "grad_norm": 6.876741409301758,
+      "learning_rate": 2e-05,
+      "loss": 0.4723,
+      "step": 62
+    },
+    {
+      "epoch": 0.64,
+      "grad_norm": 8.655380249023438,
+      "learning_rate": 2e-05,
+      "loss": 1.6293,
+      "step": 64
+    },
+    {
+      "epoch": 0.66,
+      "grad_norm": 3.3928043842315674,
+      "learning_rate": 2e-05,
+      "loss": 0.2042,
+      "step": 66
+    },
+    {
+      "epoch": 0.68,
+      "grad_norm": 0.3449620306491852,
+      "learning_rate": 2e-05,
+      "loss": 0.0418,
+      "step": 68
+    },
+    {
+      "epoch": 0.7,
+      "grad_norm": 0.8447442650794983,
+      "learning_rate": 2e-05,
+      "loss": 0.0398,
+      "step": 70
+    },
+    {
+      "epoch": 0.72,
+      "grad_norm": 5.244335651397705,
+      "learning_rate": 2e-05,
+      "loss": 0.3679,
+      "step": 72
+    },
+    {
+      "epoch": 0.74,
+      "grad_norm": 1.5702224969863892,
+      "learning_rate": 2e-05,
+      "loss": 0.0593,
+      "step": 74
+    },
+    {
+      "epoch": 0.76,
+      "grad_norm": 1.0168324708938599,
+      "learning_rate": 2e-05,
+      "loss": 0.0725,
+      "step": 76
+    },
+    {
+      "epoch": 0.78,
+      "grad_norm": 0.3886169493198395,
+      "learning_rate": 2e-05,
+      "loss": 0.2775,
+      "step": 78
+    },
+    {
+      "epoch": 0.8,
+      "grad_norm": 0.5766316652297974,
+      "learning_rate": 2e-05,
+      "loss": 0.0425,
+      "step": 80
+    },
+    {
+      "epoch": 0.82,
+      "grad_norm": 7.655923843383789,
+      "learning_rate": 2e-05,
+      "loss": 1.0432,
+      "step": 82
+    },
+    {
+      "epoch": 0.84,
+      "grad_norm": 1.8649804592132568,
+      "learning_rate": 2e-05,
+      "loss": 0.0783,
+      "step": 84
+    },
+    {
+      "epoch": 0.86,
+      "grad_norm": 0.19814282655715942,
+      "learning_rate": 2e-05,
+      "loss": 0.0133,
+      "step": 86
+    },
+    {
+      "epoch": 0.88,
+      "grad_norm": 2.5453948974609375,
+      "learning_rate": 2e-05,
+      "loss": 0.1318,
+      "step": 88
+    },
+    {
+      "epoch": 0.9,
+      "grad_norm": 1.7596789598464966,
+      "learning_rate": 2e-05,
+      "loss": 0.544,
+      "step": 90
+    },
+    {
+      "epoch": 0.92,
+      "grad_norm": 8.948413848876953,
+      "learning_rate": 2e-05,
+      "loss": 1.3617,
+      "step": 92
+    },
+    {
+      "epoch": 0.94,
+      "grad_norm": 0.24089735746383667,
+      "learning_rate": 2e-05,
+      "loss": 0.0106,
+      "step": 94
+    },
+    {
+      "epoch": 0.96,
+      "grad_norm": 1.874826192855835,
+      "learning_rate": 2e-05,
+      "loss": 0.2241,
+      "step": 96
+    },
+    {
+      "epoch": 0.98,
+      "grad_norm": 0.6936571598052979,
+      "learning_rate": 2e-05,
+      "loss": 0.0262,
+      "step": 98
+    },
+    {
+      "epoch": 1.0,
+      "grad_norm": 0.33103814721107483,
+      "learning_rate": 2e-05,
+      "loss": 0.0278,
+      "step": 100
+    },
+    {
+      "epoch": 1.0,
+      "step": 100,
+      "total_flos": 5020151544020992.0,
+      "train_loss": 0.48555011510849,
+      "train_runtime": 169.5382,
+      "train_samples_per_second": 2.359,
+      "train_steps_per_second": 0.59
+    }
+  ],
+  "logging_steps": 2,
+  "max_steps": 100,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 500,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": false,
+        "should_training_stop": false
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 5020151544020992.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}