thkim0305 commited on
Commit
5f65eef
·
verified ·
1 Parent(s): a757b8c

Upload folder using huggingface_hub

Browse files
This view is limited to 50 files because it contains too many changes.   See raw diff
Files changed (50) hide show
  1. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth +3 -0
  2. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth +3 -0
  3. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth +3 -0
  4. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth +3 -0
  5. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth +3 -0
  6. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth +3 -0
  7. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth +3 -0
  8. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth +3 -0
  9. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json +392 -0
  10. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth +3 -0
  11. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth +3 -0
  12. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth +3 -0
  13. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth +3 -0
  14. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth +3 -0
  15. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth +3 -0
  16. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth +3 -0
  17. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth +3 -0
  18. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json +392 -0
  19. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth +3 -0
  20. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth +3 -0
  21. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth +3 -0
  22. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth +3 -0
  23. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth +3 -0
  24. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth +3 -0
  25. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth +3 -0
  26. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth +3 -0
  27. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json +392 -0
  28. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth +3 -0
  29. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth +3 -0
  30. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth +3 -0
  31. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth +3 -0
  32. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth +3 -0
  33. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth +3 -0
  34. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth +3 -0
  35. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth +3 -0
  36. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json +392 -0
  37. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth +3 -0
  38. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth +3 -0
  39. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth +3 -0
  40. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth +3 -0
  41. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth +3 -0
  42. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth +3 -0
  43. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth +3 -0
  44. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth +3 -0
  45. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json +392 -0
  46. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth +3 -0
  47. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth +3 -0
  48. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth +3 -0
  49. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth +3 -0
  50. client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth +3 -0
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round10.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:0d45b60c3377d2b233fb79150e4c6377564419264b3e0dd37b143f4dea3eeaca
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d914723f4cff80df3e933a3d63ed043f08536b104877dd055d63014f403de212
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round15.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:af842a7f9e95aa8f8e6a126bead7c15243b1d3dd91e7dd5062278c436136c8d6
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round17.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b7608843e5153961ba8c3edc1f197c34164f0867a809f0207e1008fecf20a06a
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:44243ca3716547c0c21a27577345ceaeb9e623be86ced11e1044d7ab6fa8a7e7
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round20.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ea2112c3636246412e69c2fcd14b814cd7b68334b9c3f8bbddb5b8208429e04a
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1b8c7efe671d91efab26d3fbdc245783660112915cbb1d22a44ed154e94e2748
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_client_model_round7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:291f338c17fa2a84cb07d5db8619e1333d6e2f8944c902a3261c101b354ba731
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/0_trainer_state.json ADDED
@@ -0,0 +1,392 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 1.0,
5
+ "eval_steps": 500,
6
+ "global_step": 100,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.02,
13
+ "grad_norm": 10.044001579284668,
14
+ "learning_rate": 2e-05,
15
+ "loss": 1.0811,
16
+ "step": 2
17
+ },
18
+ {
19
+ "epoch": 0.04,
20
+ "grad_norm": 5.708449363708496,
21
+ "learning_rate": 2e-05,
22
+ "loss": 1.0935,
23
+ "step": 4
24
+ },
25
+ {
26
+ "epoch": 0.06,
27
+ "grad_norm": 6.164332389831543,
28
+ "learning_rate": 2e-05,
29
+ "loss": 0.2887,
30
+ "step": 6
31
+ },
32
+ {
33
+ "epoch": 0.08,
34
+ "grad_norm": 0.7329395413398743,
35
+ "learning_rate": 2e-05,
36
+ "loss": 0.0654,
37
+ "step": 8
38
+ },
39
+ {
40
+ "epoch": 0.1,
41
+ "grad_norm": 3.7981791496276855,
42
+ "learning_rate": 2e-05,
43
+ "loss": 0.5255,
44
+ "step": 10
45
+ },
46
+ {
47
+ "epoch": 0.12,
48
+ "grad_norm": 10.993570327758789,
49
+ "learning_rate": 2e-05,
50
+ "loss": 0.7257,
51
+ "step": 12
52
+ },
53
+ {
54
+ "epoch": 0.14,
55
+ "grad_norm": 6.135924816131592,
56
+ "learning_rate": 2e-05,
57
+ "loss": 0.3716,
58
+ "step": 14
59
+ },
60
+ {
61
+ "epoch": 0.16,
62
+ "grad_norm": 10.345540046691895,
63
+ "learning_rate": 2e-05,
64
+ "loss": 1.132,
65
+ "step": 16
66
+ },
67
+ {
68
+ "epoch": 0.18,
69
+ "grad_norm": 15.495490074157715,
70
+ "learning_rate": 2e-05,
71
+ "loss": 1.7626,
72
+ "step": 18
73
+ },
74
+ {
75
+ "epoch": 0.2,
76
+ "grad_norm": 9.968962669372559,
77
+ "learning_rate": 2e-05,
78
+ "loss": 1.0484,
79
+ "step": 20
80
+ },
81
+ {
82
+ "epoch": 0.22,
83
+ "grad_norm": 0.6021881103515625,
84
+ "learning_rate": 2e-05,
85
+ "loss": 0.0195,
86
+ "step": 22
87
+ },
88
+ {
89
+ "epoch": 0.24,
90
+ "grad_norm": 10.388916015625,
91
+ "learning_rate": 2e-05,
92
+ "loss": 0.813,
93
+ "step": 24
94
+ },
95
+ {
96
+ "epoch": 0.26,
97
+ "grad_norm": 6.319398880004883,
98
+ "learning_rate": 2e-05,
99
+ "loss": 0.2256,
100
+ "step": 26
101
+ },
102
+ {
103
+ "epoch": 0.28,
104
+ "grad_norm": 0.6123639345169067,
105
+ "learning_rate": 2e-05,
106
+ "loss": 0.1623,
107
+ "step": 28
108
+ },
109
+ {
110
+ "epoch": 0.3,
111
+ "grad_norm": 2.0997838973999023,
112
+ "learning_rate": 2e-05,
113
+ "loss": 0.8552,
114
+ "step": 30
115
+ },
116
+ {
117
+ "epoch": 0.32,
118
+ "grad_norm": 17.500144958496094,
119
+ "learning_rate": 2e-05,
120
+ "loss": 1.1279,
121
+ "step": 32
122
+ },
123
+ {
124
+ "epoch": 0.34,
125
+ "grad_norm": 5.181461334228516,
126
+ "learning_rate": 2e-05,
127
+ "loss": 0.2672,
128
+ "step": 34
129
+ },
130
+ {
131
+ "epoch": 0.36,
132
+ "grad_norm": 10.427047729492188,
133
+ "learning_rate": 2e-05,
134
+ "loss": 0.5116,
135
+ "step": 36
136
+ },
137
+ {
138
+ "epoch": 0.38,
139
+ "grad_norm": 4.001834392547607,
140
+ "learning_rate": 2e-05,
141
+ "loss": 0.8254,
142
+ "step": 38
143
+ },
144
+ {
145
+ "epoch": 0.4,
146
+ "grad_norm": 4.918587684631348,
147
+ "learning_rate": 2e-05,
148
+ "loss": 0.2577,
149
+ "step": 40
150
+ },
151
+ {
152
+ "epoch": 0.42,
153
+ "grad_norm": 1.5070538520812988,
154
+ "learning_rate": 2e-05,
155
+ "loss": 0.0786,
156
+ "step": 42
157
+ },
158
+ {
159
+ "epoch": 0.44,
160
+ "grad_norm": 8.23068904876709,
161
+ "learning_rate": 2e-05,
162
+ "loss": 1.336,
163
+ "step": 44
164
+ },
165
+ {
166
+ "epoch": 0.46,
167
+ "grad_norm": 0.4960165321826935,
168
+ "learning_rate": 2e-05,
169
+ "loss": 0.2786,
170
+ "step": 46
171
+ },
172
+ {
173
+ "epoch": 0.48,
174
+ "grad_norm": 14.814494132995605,
175
+ "learning_rate": 2e-05,
176
+ "loss": 1.3887,
177
+ "step": 48
178
+ },
179
+ {
180
+ "epoch": 0.5,
181
+ "grad_norm": 2.1397197246551514,
182
+ "learning_rate": 2e-05,
183
+ "loss": 0.1331,
184
+ "step": 50
185
+ },
186
+ {
187
+ "epoch": 0.52,
188
+ "grad_norm": 11.526412963867188,
189
+ "learning_rate": 2e-05,
190
+ "loss": 0.4426,
191
+ "step": 52
192
+ },
193
+ {
194
+ "epoch": 0.54,
195
+ "grad_norm": 2.414958953857422,
196
+ "learning_rate": 2e-05,
197
+ "loss": 0.2413,
198
+ "step": 54
199
+ },
200
+ {
201
+ "epoch": 0.56,
202
+ "grad_norm": 4.718291282653809,
203
+ "learning_rate": 2e-05,
204
+ "loss": 0.1895,
205
+ "step": 56
206
+ },
207
+ {
208
+ "epoch": 0.58,
209
+ "grad_norm": 12.340472221374512,
210
+ "learning_rate": 2e-05,
211
+ "loss": 1.9869,
212
+ "step": 58
213
+ },
214
+ {
215
+ "epoch": 0.6,
216
+ "grad_norm": 4.813292503356934,
217
+ "learning_rate": 2e-05,
218
+ "loss": 0.1454,
219
+ "step": 60
220
+ },
221
+ {
222
+ "epoch": 0.62,
223
+ "grad_norm": 2.3212244510650635,
224
+ "learning_rate": 2e-05,
225
+ "loss": 0.4304,
226
+ "step": 62
227
+ },
228
+ {
229
+ "epoch": 0.64,
230
+ "grad_norm": 3.7471375465393066,
231
+ "learning_rate": 2e-05,
232
+ "loss": 0.1793,
233
+ "step": 64
234
+ },
235
+ {
236
+ "epoch": 0.66,
237
+ "grad_norm": 0.6779154539108276,
238
+ "learning_rate": 2e-05,
239
+ "loss": 0.1912,
240
+ "step": 66
241
+ },
242
+ {
243
+ "epoch": 0.68,
244
+ "grad_norm": 5.2416229248046875,
245
+ "learning_rate": 2e-05,
246
+ "loss": 0.7198,
247
+ "step": 68
248
+ },
249
+ {
250
+ "epoch": 0.7,
251
+ "grad_norm": 12.138299942016602,
252
+ "learning_rate": 2e-05,
253
+ "loss": 0.6979,
254
+ "step": 70
255
+ },
256
+ {
257
+ "epoch": 0.72,
258
+ "grad_norm": 8.756268501281738,
259
+ "learning_rate": 2e-05,
260
+ "loss": 0.3211,
261
+ "step": 72
262
+ },
263
+ {
264
+ "epoch": 0.74,
265
+ "grad_norm": 12.594830513000488,
266
+ "learning_rate": 2e-05,
267
+ "loss": 2.7682,
268
+ "step": 74
269
+ },
270
+ {
271
+ "epoch": 0.76,
272
+ "grad_norm": 5.925561904907227,
273
+ "learning_rate": 2e-05,
274
+ "loss": 0.7106,
275
+ "step": 76
276
+ },
277
+ {
278
+ "epoch": 0.78,
279
+ "grad_norm": 0.4183579981327057,
280
+ "learning_rate": 2e-05,
281
+ "loss": 0.9514,
282
+ "step": 78
283
+ },
284
+ {
285
+ "epoch": 0.8,
286
+ "grad_norm": 6.107239246368408,
287
+ "learning_rate": 2e-05,
288
+ "loss": 1.1018,
289
+ "step": 80
290
+ },
291
+ {
292
+ "epoch": 0.82,
293
+ "grad_norm": 6.866254806518555,
294
+ "learning_rate": 2e-05,
295
+ "loss": 0.6412,
296
+ "step": 82
297
+ },
298
+ {
299
+ "epoch": 0.84,
300
+ "grad_norm": 13.476661682128906,
301
+ "learning_rate": 2e-05,
302
+ "loss": 0.9713,
303
+ "step": 84
304
+ },
305
+ {
306
+ "epoch": 0.86,
307
+ "grad_norm": 0.8483925461769104,
308
+ "learning_rate": 2e-05,
309
+ "loss": 0.1767,
310
+ "step": 86
311
+ },
312
+ {
313
+ "epoch": 0.88,
314
+ "grad_norm": 9.308485984802246,
315
+ "learning_rate": 2e-05,
316
+ "loss": 1.2229,
317
+ "step": 88
318
+ },
319
+ {
320
+ "epoch": 0.9,
321
+ "grad_norm": 4.227640151977539,
322
+ "learning_rate": 2e-05,
323
+ "loss": 0.4012,
324
+ "step": 90
325
+ },
326
+ {
327
+ "epoch": 0.92,
328
+ "grad_norm": 0.6228331923484802,
329
+ "learning_rate": 2e-05,
330
+ "loss": 0.9619,
331
+ "step": 92
332
+ },
333
+ {
334
+ "epoch": 0.94,
335
+ "grad_norm": 9.9182767868042,
336
+ "learning_rate": 2e-05,
337
+ "loss": 2.1173,
338
+ "step": 94
339
+ },
340
+ {
341
+ "epoch": 0.96,
342
+ "grad_norm": 3.4465816020965576,
343
+ "learning_rate": 2e-05,
344
+ "loss": 1.605,
345
+ "step": 96
346
+ },
347
+ {
348
+ "epoch": 0.98,
349
+ "grad_norm": 12.18694019317627,
350
+ "learning_rate": 2e-05,
351
+ "loss": 2.0198,
352
+ "step": 98
353
+ },
354
+ {
355
+ "epoch": 1.0,
356
+ "grad_norm": 1.0387413501739502,
357
+ "learning_rate": 2e-05,
358
+ "loss": 0.2066,
359
+ "step": 100
360
+ },
361
+ {
362
+ "epoch": 1.0,
363
+ "step": 100,
364
+ "total_flos": 2053257199353856.0,
365
+ "train_loss": 0.755521228313446,
366
+ "train_runtime": 101.7028,
367
+ "train_samples_per_second": 3.933,
368
+ "train_steps_per_second": 0.983
369
+ }
370
+ ],
371
+ "logging_steps": 2,
372
+ "max_steps": 100,
373
+ "num_input_tokens_seen": 0,
374
+ "num_train_epochs": 1,
375
+ "save_steps": 500,
376
+ "stateful_callbacks": {
377
+ "TrainerControl": {
378
+ "args": {
379
+ "should_epoch_stop": false,
380
+ "should_evaluate": false,
381
+ "should_log": false,
382
+ "should_save": false,
383
+ "should_training_stop": false
384
+ },
385
+ "attributes": {}
386
+ }
387
+ },
388
+ "total_flos": 2053257199353856.0,
389
+ "train_batch_size": 1,
390
+ "trial_name": null,
391
+ "trial_params": null
392
+ }
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round10.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bf0af71fd4d6dc12cffd01523df4bbc58a571c0978ca4c7d7571aa24bdeb8ee5
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1744e2ae1501079d9883e05d012fe5bf1d7a4f72884671f31cc3a124d304ec02
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round15.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c105cc5e4f96671ece69b8f9d708aa9a805d9a65a48b0865cc54c7d7fc57fdbf
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round17.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ff0a15058359453e6d079b47f6e618fcf9720db11d691877dbe2140c6a15be3c
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ec3b7d6ad6a720fb656550c2f62665f56860a6f8a8288ff1f826b9dceaed89e
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round20.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:59993217d193880b0fc5548c04f04df882e9949136543c7bbede6e4dcfb2687c
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:009defbeb3403fee4f1ed358ade2be16ece2a7c97a4f95464240ea4cf6c2e3d8
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_client_model_round7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c0be9e6109694003abb07f24a9bb7b03b63669d481039c2e19e40b6296e98501
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/1_trainer_state.json ADDED
@@ -0,0 +1,392 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 1.0,
5
+ "eval_steps": 500,
6
+ "global_step": 100,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.02,
13
+ "grad_norm": 0.02960115671157837,
14
+ "learning_rate": 2e-05,
15
+ "loss": 0.0045,
16
+ "step": 2
17
+ },
18
+ {
19
+ "epoch": 0.04,
20
+ "grad_norm": 0.09186350554227829,
21
+ "learning_rate": 2e-05,
22
+ "loss": 0.0653,
23
+ "step": 4
24
+ },
25
+ {
26
+ "epoch": 0.06,
27
+ "grad_norm": 0.16298067569732666,
28
+ "learning_rate": 2e-05,
29
+ "loss": 0.0063,
30
+ "step": 6
31
+ },
32
+ {
33
+ "epoch": 0.08,
34
+ "grad_norm": 0.09170705080032349,
35
+ "learning_rate": 2e-05,
36
+ "loss": 0.0034,
37
+ "step": 8
38
+ },
39
+ {
40
+ "epoch": 0.1,
41
+ "grad_norm": 0.06864725798368454,
42
+ "learning_rate": 2e-05,
43
+ "loss": 0.0031,
44
+ "step": 10
45
+ },
46
+ {
47
+ "epoch": 0.12,
48
+ "grad_norm": 0.19071757793426514,
49
+ "learning_rate": 2e-05,
50
+ "loss": 0.0287,
51
+ "step": 12
52
+ },
53
+ {
54
+ "epoch": 0.14,
55
+ "grad_norm": 0.17343077063560486,
56
+ "learning_rate": 2e-05,
57
+ "loss": 0.0038,
58
+ "step": 14
59
+ },
60
+ {
61
+ "epoch": 0.16,
62
+ "grad_norm": 0.2856125235557556,
63
+ "learning_rate": 2e-05,
64
+ "loss": 0.0591,
65
+ "step": 16
66
+ },
67
+ {
68
+ "epoch": 0.18,
69
+ "grad_norm": 0.04609253257513046,
70
+ "learning_rate": 2e-05,
71
+ "loss": 0.024,
72
+ "step": 18
73
+ },
74
+ {
75
+ "epoch": 0.2,
76
+ "grad_norm": 0.01641538180410862,
77
+ "learning_rate": 2e-05,
78
+ "loss": 0.001,
79
+ "step": 20
80
+ },
81
+ {
82
+ "epoch": 0.22,
83
+ "grad_norm": 0.07860966771841049,
84
+ "learning_rate": 2e-05,
85
+ "loss": 0.0113,
86
+ "step": 22
87
+ },
88
+ {
89
+ "epoch": 0.24,
90
+ "grad_norm": 0.027962295338511467,
91
+ "learning_rate": 2e-05,
92
+ "loss": 0.0011,
93
+ "step": 24
94
+ },
95
+ {
96
+ "epoch": 0.26,
97
+ "grad_norm": 1.3819764852523804,
98
+ "learning_rate": 2e-05,
99
+ "loss": 0.0356,
100
+ "step": 26
101
+ },
102
+ {
103
+ "epoch": 0.28,
104
+ "grad_norm": 0.01595745049417019,
105
+ "learning_rate": 2e-05,
106
+ "loss": 0.0008,
107
+ "step": 28
108
+ },
109
+ {
110
+ "epoch": 0.3,
111
+ "grad_norm": 0.12123987823724747,
112
+ "learning_rate": 2e-05,
113
+ "loss": 0.004,
114
+ "step": 30
115
+ },
116
+ {
117
+ "epoch": 0.32,
118
+ "grad_norm": 0.7385960221290588,
119
+ "learning_rate": 2e-05,
120
+ "loss": 0.1151,
121
+ "step": 32
122
+ },
123
+ {
124
+ "epoch": 0.34,
125
+ "grad_norm": 5.604942798614502,
126
+ "learning_rate": 2e-05,
127
+ "loss": 0.1588,
128
+ "step": 34
129
+ },
130
+ {
131
+ "epoch": 0.36,
132
+ "grad_norm": 0.005619928240776062,
133
+ "learning_rate": 2e-05,
134
+ "loss": 0.0003,
135
+ "step": 36
136
+ },
137
+ {
138
+ "epoch": 0.38,
139
+ "grad_norm": 0.027213625609874725,
140
+ "learning_rate": 2e-05,
141
+ "loss": 0.0016,
142
+ "step": 38
143
+ },
144
+ {
145
+ "epoch": 0.4,
146
+ "grad_norm": 4.135501384735107,
147
+ "learning_rate": 2e-05,
148
+ "loss": 0.0975,
149
+ "step": 40
150
+ },
151
+ {
152
+ "epoch": 0.42,
153
+ "grad_norm": 0.00542866624891758,
154
+ "learning_rate": 2e-05,
155
+ "loss": 0.0006,
156
+ "step": 42
157
+ },
158
+ {
159
+ "epoch": 0.44,
160
+ "grad_norm": 7.716265678405762,
161
+ "learning_rate": 2e-05,
162
+ "loss": 0.3077,
163
+ "step": 44
164
+ },
165
+ {
166
+ "epoch": 0.46,
167
+ "grad_norm": 0.01432905811816454,
168
+ "learning_rate": 2e-05,
169
+ "loss": 0.0009,
170
+ "step": 46
171
+ },
172
+ {
173
+ "epoch": 0.48,
174
+ "grad_norm": 0.3185250163078308,
175
+ "learning_rate": 2e-05,
176
+ "loss": 0.004,
177
+ "step": 48
178
+ },
179
+ {
180
+ "epoch": 0.5,
181
+ "grad_norm": 0.00824733730405569,
182
+ "learning_rate": 2e-05,
183
+ "loss": 0.0022,
184
+ "step": 50
185
+ },
186
+ {
187
+ "epoch": 0.52,
188
+ "grad_norm": 0.006523216143250465,
189
+ "learning_rate": 2e-05,
190
+ "loss": 0.0265,
191
+ "step": 52
192
+ },
193
+ {
194
+ "epoch": 0.54,
195
+ "grad_norm": 0.3274787664413452,
196
+ "learning_rate": 2e-05,
197
+ "loss": 0.0094,
198
+ "step": 54
199
+ },
200
+ {
201
+ "epoch": 0.56,
202
+ "grad_norm": 0.012340985238552094,
203
+ "learning_rate": 2e-05,
204
+ "loss": 0.0011,
205
+ "step": 56
206
+ },
207
+ {
208
+ "epoch": 0.58,
209
+ "grad_norm": 0.01708126626908779,
210
+ "learning_rate": 2e-05,
211
+ "loss": 0.6785,
212
+ "step": 58
213
+ },
214
+ {
215
+ "epoch": 0.6,
216
+ "grad_norm": 0.005370909348130226,
217
+ "learning_rate": 2e-05,
218
+ "loss": 0.2075,
219
+ "step": 60
220
+ },
221
+ {
222
+ "epoch": 0.62,
223
+ "grad_norm": 0.009917240589857101,
224
+ "learning_rate": 2e-05,
225
+ "loss": 0.0011,
226
+ "step": 62
227
+ },
228
+ {
229
+ "epoch": 0.64,
230
+ "grad_norm": 0.015617340803146362,
231
+ "learning_rate": 2e-05,
232
+ "loss": 0.0178,
233
+ "step": 64
234
+ },
235
+ {
236
+ "epoch": 0.66,
237
+ "grad_norm": 0.04841961711645126,
238
+ "learning_rate": 2e-05,
239
+ "loss": 0.0013,
240
+ "step": 66
241
+ },
242
+ {
243
+ "epoch": 0.68,
244
+ "grad_norm": 2.055001735687256,
245
+ "learning_rate": 2e-05,
246
+ "loss": 0.0487,
247
+ "step": 68
248
+ },
249
+ {
250
+ "epoch": 0.7,
251
+ "grad_norm": 0.010724334977567196,
252
+ "learning_rate": 2e-05,
253
+ "loss": 0.0007,
254
+ "step": 70
255
+ },
256
+ {
257
+ "epoch": 0.72,
258
+ "grad_norm": 0.1112007200717926,
259
+ "learning_rate": 2e-05,
260
+ "loss": 0.0032,
261
+ "step": 72
262
+ },
263
+ {
264
+ "epoch": 0.74,
265
+ "grad_norm": 0.5635241270065308,
266
+ "learning_rate": 2e-05,
267
+ "loss": 0.0103,
268
+ "step": 74
269
+ },
270
+ {
271
+ "epoch": 0.76,
272
+ "grad_norm": 0.024422986432909966,
273
+ "learning_rate": 2e-05,
274
+ "loss": 0.0009,
275
+ "step": 76
276
+ },
277
+ {
278
+ "epoch": 0.78,
279
+ "grad_norm": 9.877224922180176,
280
+ "learning_rate": 2e-05,
281
+ "loss": 0.4788,
282
+ "step": 78
283
+ },
284
+ {
285
+ "epoch": 0.8,
286
+ "grad_norm": 0.025433897972106934,
287
+ "learning_rate": 2e-05,
288
+ "loss": 0.0017,
289
+ "step": 80
290
+ },
291
+ {
292
+ "epoch": 0.82,
293
+ "grad_norm": 0.07279513031244278,
294
+ "learning_rate": 2e-05,
295
+ "loss": 0.0021,
296
+ "step": 82
297
+ },
298
+ {
299
+ "epoch": 0.84,
300
+ "grad_norm": 0.003639621427282691,
301
+ "learning_rate": 2e-05,
302
+ "loss": 0.0004,
303
+ "step": 84
304
+ },
305
+ {
306
+ "epoch": 0.86,
307
+ "grad_norm": 0.021531013771891594,
308
+ "learning_rate": 2e-05,
309
+ "loss": 1.1024,
310
+ "step": 86
311
+ },
312
+ {
313
+ "epoch": 0.88,
314
+ "grad_norm": 0.041703518480062485,
315
+ "learning_rate": 2e-05,
316
+ "loss": 0.0053,
317
+ "step": 88
318
+ },
319
+ {
320
+ "epoch": 0.9,
321
+ "grad_norm": 1.0306504964828491,
322
+ "learning_rate": 2e-05,
323
+ "loss": 0.0258,
324
+ "step": 90
325
+ },
326
+ {
327
+ "epoch": 0.92,
328
+ "grad_norm": 0.19764401018619537,
329
+ "learning_rate": 2e-05,
330
+ "loss": 0.0066,
331
+ "step": 92
332
+ },
333
+ {
334
+ "epoch": 0.94,
335
+ "grad_norm": 0.1398591846227646,
336
+ "learning_rate": 2e-05,
337
+ "loss": 0.0036,
338
+ "step": 94
339
+ },
340
+ {
341
+ "epoch": 0.96,
342
+ "grad_norm": 0.9593293070793152,
343
+ "learning_rate": 2e-05,
344
+ "loss": 0.0234,
345
+ "step": 96
346
+ },
347
+ {
348
+ "epoch": 0.98,
349
+ "grad_norm": 0.013573438860476017,
350
+ "learning_rate": 2e-05,
351
+ "loss": 0.0012,
352
+ "step": 98
353
+ },
354
+ {
355
+ "epoch": 1.0,
356
+ "grad_norm": 0.1273522675037384,
357
+ "learning_rate": 2e-05,
358
+ "loss": 0.003,
359
+ "step": 100
360
+ },
361
+ {
362
+ "epoch": 1.0,
363
+ "step": 100,
364
+ "total_flos": 2069634366832640.0,
365
+ "train_loss": 0.07204394578933716,
366
+ "train_runtime": 101.8827,
367
+ "train_samples_per_second": 3.926,
368
+ "train_steps_per_second": 0.982
369
+ }
370
+ ],
371
+ "logging_steps": 2,
372
+ "max_steps": 100,
373
+ "num_input_tokens_seen": 0,
374
+ "num_train_epochs": 1,
375
+ "save_steps": 500,
376
+ "stateful_callbacks": {
377
+ "TrainerControl": {
378
+ "args": {
379
+ "should_epoch_stop": false,
380
+ "should_evaluate": false,
381
+ "should_log": false,
382
+ "should_save": false,
383
+ "should_training_stop": false
384
+ },
385
+ "attributes": {}
386
+ }
387
+ },
388
+ "total_flos": 2069634366832640.0,
389
+ "train_batch_size": 1,
390
+ "trial_name": null,
391
+ "trial_params": null
392
+ }
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round10.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:2cd1af5c7ca40ed6a79af8c9cb6555f78589b0a624a98c6483565f76f8b3c61d
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8e8db792016ac55ff57bd23f8349e5df6ceb14624e1d9a29105c81ed7d7ed4ed
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round15.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:19524fc38015f33290f292a4f522b4d36c1adcb621ac578bb561050e13fefe91
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round17.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a874d7d599aad5817e884d0953977542823faf1f01d25101400d3430928ff9c
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:33baa7cb694bc7b9f9162330827005d81631575ea2e90cd715b6d42b50f475da
3
+ size 395786922
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round20.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cb3be0c563957173b91a7ec86c8723fdb23acdadf722cf602bc3483c5d0b11ae
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:698b9afe0148feecdd937cdc3624b381dfe48c2a3a8083e2dfa29bfedd728c42
3
+ size 395786922
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_client_model_round7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:633deab3c32fa4046bc1e4c0691c368ae244561291e78cbdba6062fdf7b77f16
3
+ size 395786922
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/2_trainer_state.json ADDED
@@ -0,0 +1,392 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 1.0,
5
+ "eval_steps": 500,
6
+ "global_step": 100,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.02,
13
+ "grad_norm": 1.765663981437683,
14
+ "learning_rate": 2e-05,
15
+ "loss": 0.1386,
16
+ "step": 2
17
+ },
18
+ {
19
+ "epoch": 0.04,
20
+ "grad_norm": 4.676456928253174,
21
+ "learning_rate": 2e-05,
22
+ "loss": 1.1946,
23
+ "step": 4
24
+ },
25
+ {
26
+ "epoch": 0.06,
27
+ "grad_norm": 2.184738874435425,
28
+ "learning_rate": 2e-05,
29
+ "loss": 0.9233,
30
+ "step": 6
31
+ },
32
+ {
33
+ "epoch": 0.08,
34
+ "grad_norm": 4.627042770385742,
35
+ "learning_rate": 2e-05,
36
+ "loss": 1.1015,
37
+ "step": 8
38
+ },
39
+ {
40
+ "epoch": 0.1,
41
+ "grad_norm": 5.915734767913818,
42
+ "learning_rate": 2e-05,
43
+ "loss": 0.663,
44
+ "step": 10
45
+ },
46
+ {
47
+ "epoch": 0.12,
48
+ "grad_norm": 6.462231636047363,
49
+ "learning_rate": 2e-05,
50
+ "loss": 1.0081,
51
+ "step": 12
52
+ },
53
+ {
54
+ "epoch": 0.14,
55
+ "grad_norm": 0.35552021861076355,
56
+ "learning_rate": 2e-05,
57
+ "loss": 0.3945,
58
+ "step": 14
59
+ },
60
+ {
61
+ "epoch": 0.16,
62
+ "grad_norm": 5.338579177856445,
63
+ "learning_rate": 2e-05,
64
+ "loss": 0.6744,
65
+ "step": 16
66
+ },
67
+ {
68
+ "epoch": 0.18,
69
+ "grad_norm": 1.783922791481018,
70
+ "learning_rate": 2e-05,
71
+ "loss": 0.4461,
72
+ "step": 18
73
+ },
74
+ {
75
+ "epoch": 0.2,
76
+ "grad_norm": 3.711841106414795,
77
+ "learning_rate": 2e-05,
78
+ "loss": 1.4118,
79
+ "step": 20
80
+ },
81
+ {
82
+ "epoch": 0.22,
83
+ "grad_norm": 0.9069531559944153,
84
+ "learning_rate": 2e-05,
85
+ "loss": 0.176,
86
+ "step": 22
87
+ },
88
+ {
89
+ "epoch": 0.24,
90
+ "grad_norm": 7.7085490226745605,
91
+ "learning_rate": 2e-05,
92
+ "loss": 0.5315,
93
+ "step": 24
94
+ },
95
+ {
96
+ "epoch": 0.26,
97
+ "grad_norm": 10.80366039276123,
98
+ "learning_rate": 2e-05,
99
+ "loss": 1.0346,
100
+ "step": 26
101
+ },
102
+ {
103
+ "epoch": 0.28,
104
+ "grad_norm": 6.07086181640625,
105
+ "learning_rate": 2e-05,
106
+ "loss": 0.3799,
107
+ "step": 28
108
+ },
109
+ {
110
+ "epoch": 0.3,
111
+ "grad_norm": 1.8229091167449951,
112
+ "learning_rate": 2e-05,
113
+ "loss": 0.1932,
114
+ "step": 30
115
+ },
116
+ {
117
+ "epoch": 0.32,
118
+ "grad_norm": 5.323653697967529,
119
+ "learning_rate": 2e-05,
120
+ "loss": 0.3793,
121
+ "step": 32
122
+ },
123
+ {
124
+ "epoch": 0.34,
125
+ "grad_norm": 4.461909770965576,
126
+ "learning_rate": 2e-05,
127
+ "loss": 0.6431,
128
+ "step": 34
129
+ },
130
+ {
131
+ "epoch": 0.36,
132
+ "grad_norm": 1.6867424249649048,
133
+ "learning_rate": 2e-05,
134
+ "loss": 0.1087,
135
+ "step": 36
136
+ },
137
+ {
138
+ "epoch": 0.38,
139
+ "grad_norm": 3.3799192905426025,
140
+ "learning_rate": 2e-05,
141
+ "loss": 0.6874,
142
+ "step": 38
143
+ },
144
+ {
145
+ "epoch": 0.4,
146
+ "grad_norm": 7.544458389282227,
147
+ "learning_rate": 2e-05,
148
+ "loss": 0.6817,
149
+ "step": 40
150
+ },
151
+ {
152
+ "epoch": 0.42,
153
+ "grad_norm": 2.379153251647949,
154
+ "learning_rate": 2e-05,
155
+ "loss": 0.6136,
156
+ "step": 42
157
+ },
158
+ {
159
+ "epoch": 0.44,
160
+ "grad_norm": 2.67325758934021,
161
+ "learning_rate": 2e-05,
162
+ "loss": 0.1987,
163
+ "step": 44
164
+ },
165
+ {
166
+ "epoch": 0.46,
167
+ "grad_norm": 1.6791194677352905,
168
+ "learning_rate": 2e-05,
169
+ "loss": 0.3694,
170
+ "step": 46
171
+ },
172
+ {
173
+ "epoch": 0.48,
174
+ "grad_norm": 0.7336940169334412,
175
+ "learning_rate": 2e-05,
176
+ "loss": 0.3386,
177
+ "step": 48
178
+ },
179
+ {
180
+ "epoch": 0.5,
181
+ "grad_norm": 2.389741897583008,
182
+ "learning_rate": 2e-05,
183
+ "loss": 0.1778,
184
+ "step": 50
185
+ },
186
+ {
187
+ "epoch": 0.52,
188
+ "grad_norm": 7.384143829345703,
189
+ "learning_rate": 2e-05,
190
+ "loss": 0.7985,
191
+ "step": 52
192
+ },
193
+ {
194
+ "epoch": 0.54,
195
+ "grad_norm": 7.0400872230529785,
196
+ "learning_rate": 2e-05,
197
+ "loss": 0.6097,
198
+ "step": 54
199
+ },
200
+ {
201
+ "epoch": 0.56,
202
+ "grad_norm": 8.808256149291992,
203
+ "learning_rate": 2e-05,
204
+ "loss": 0.7863,
205
+ "step": 56
206
+ },
207
+ {
208
+ "epoch": 0.58,
209
+ "grad_norm": 3.6529507637023926,
210
+ "learning_rate": 2e-05,
211
+ "loss": 0.1368,
212
+ "step": 58
213
+ },
214
+ {
215
+ "epoch": 0.6,
216
+ "grad_norm": 0.17764848470687866,
217
+ "learning_rate": 2e-05,
218
+ "loss": 0.1653,
219
+ "step": 60
220
+ },
221
+ {
222
+ "epoch": 0.62,
223
+ "grad_norm": 1.8120052814483643,
224
+ "learning_rate": 2e-05,
225
+ "loss": 0.7154,
226
+ "step": 62
227
+ },
228
+ {
229
+ "epoch": 0.64,
230
+ "grad_norm": 3.756453037261963,
231
+ "learning_rate": 2e-05,
232
+ "loss": 0.5272,
233
+ "step": 64
234
+ },
235
+ {
236
+ "epoch": 0.66,
237
+ "grad_norm": 3.120513677597046,
238
+ "learning_rate": 2e-05,
239
+ "loss": 0.1667,
240
+ "step": 66
241
+ },
242
+ {
243
+ "epoch": 0.68,
244
+ "grad_norm": 4.981636047363281,
245
+ "learning_rate": 2e-05,
246
+ "loss": 0.3927,
247
+ "step": 68
248
+ },
249
+ {
250
+ "epoch": 0.7,
251
+ "grad_norm": 5.101208209991455,
252
+ "learning_rate": 2e-05,
253
+ "loss": 0.6766,
254
+ "step": 70
255
+ },
256
+ {
257
+ "epoch": 0.72,
258
+ "grad_norm": 1.0963003635406494,
259
+ "learning_rate": 2e-05,
260
+ "loss": 0.7616,
261
+ "step": 72
262
+ },
263
+ {
264
+ "epoch": 0.74,
265
+ "grad_norm": 0.7717503905296326,
266
+ "learning_rate": 2e-05,
267
+ "loss": 0.0356,
268
+ "step": 74
269
+ },
270
+ {
271
+ "epoch": 0.76,
272
+ "grad_norm": 0.8139798641204834,
273
+ "learning_rate": 2e-05,
274
+ "loss": 0.1705,
275
+ "step": 76
276
+ },
277
+ {
278
+ "epoch": 0.78,
279
+ "grad_norm": 3.033977746963501,
280
+ "learning_rate": 2e-05,
281
+ "loss": 0.1349,
282
+ "step": 78
283
+ },
284
+ {
285
+ "epoch": 0.8,
286
+ "grad_norm": 13.496634483337402,
287
+ "learning_rate": 2e-05,
288
+ "loss": 1.6624,
289
+ "step": 80
290
+ },
291
+ {
292
+ "epoch": 0.82,
293
+ "grad_norm": 13.66792106628418,
294
+ "learning_rate": 2e-05,
295
+ "loss": 2.9498,
296
+ "step": 82
297
+ },
298
+ {
299
+ "epoch": 0.84,
300
+ "grad_norm": 0.06267403066158295,
301
+ "learning_rate": 2e-05,
302
+ "loss": 0.0111,
303
+ "step": 84
304
+ },
305
+ {
306
+ "epoch": 0.86,
307
+ "grad_norm": 5.407293319702148,
308
+ "learning_rate": 2e-05,
309
+ "loss": 0.5322,
310
+ "step": 86
311
+ },
312
+ {
313
+ "epoch": 0.88,
314
+ "grad_norm": 6.728641510009766,
315
+ "learning_rate": 2e-05,
316
+ "loss": 0.6077,
317
+ "step": 88
318
+ },
319
+ {
320
+ "epoch": 0.9,
321
+ "grad_norm": 5.393722057342529,
322
+ "learning_rate": 2e-05,
323
+ "loss": 1.6041,
324
+ "step": 90
325
+ },
326
+ {
327
+ "epoch": 0.92,
328
+ "grad_norm": 0.15307481586933136,
329
+ "learning_rate": 2e-05,
330
+ "loss": 0.3262,
331
+ "step": 92
332
+ },
333
+ {
334
+ "epoch": 0.94,
335
+ "grad_norm": 6.039478302001953,
336
+ "learning_rate": 2e-05,
337
+ "loss": 0.491,
338
+ "step": 94
339
+ },
340
+ {
341
+ "epoch": 0.96,
342
+ "grad_norm": 3.6059703826904297,
343
+ "learning_rate": 2e-05,
344
+ "loss": 0.2965,
345
+ "step": 96
346
+ },
347
+ {
348
+ "epoch": 0.98,
349
+ "grad_norm": 10.516265869140625,
350
+ "learning_rate": 2e-05,
351
+ "loss": 1.887,
352
+ "step": 98
353
+ },
354
+ {
355
+ "epoch": 1.0,
356
+ "grad_norm": 2.030522584915161,
357
+ "learning_rate": 2e-05,
358
+ "loss": 0.1721,
359
+ "step": 100
360
+ },
361
+ {
362
+ "epoch": 1.0,
363
+ "step": 100,
364
+ "total_flos": 4914533793529856.0,
365
+ "train_loss": 0.6217503023147583,
366
+ "train_runtime": 169.8273,
367
+ "train_samples_per_second": 2.355,
368
+ "train_steps_per_second": 0.589
369
+ }
370
+ ],
371
+ "logging_steps": 2,
372
+ "max_steps": 100,
373
+ "num_input_tokens_seen": 0,
374
+ "num_train_epochs": 1,
375
+ "save_steps": 500,
376
+ "stateful_callbacks": {
377
+ "TrainerControl": {
378
+ "args": {
379
+ "should_epoch_stop": false,
380
+ "should_evaluate": false,
381
+ "should_log": false,
382
+ "should_save": false,
383
+ "should_training_stop": false
384
+ },
385
+ "attributes": {}
386
+ }
387
+ },
388
+ "total_flos": 4914533793529856.0,
389
+ "train_batch_size": 1,
390
+ "trial_name": null,
391
+ "trial_params": null
392
+ }
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round10.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:386840b3ff94a28b7c83212829e6accfe75f2e328a9057cab81f329994fef0f1
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:99fe6a5d31215c1d0c764a74fdc4ec5918e52a0db960e55f5fdd3fe51f8dc2d2
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round15.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:924e12ae251af8a13eb6310668a2e9dbdc07b477f9b1ade3564b97b41171fb39
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round17.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:dc65366e8dea534e149ce7cd583b6319994ed35ecc39582525344b753aeb1858
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:92099a5ace36a30279e1cc431967135db0772dc3159d7b8d231675dfd4cf88c0
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round20.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:d874d9cb32ec1bddc888cfd18e4b414cf20eff538ba46baccd8b8d0b89bc4fbd
3
+ size 184221358
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:85cba7e8c401b0d413e0b2a269065ccf0c313caf14e5148232105b7da2d2fa8f
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_client_model_round7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b6a290cb3bf0eb2e5195b515d1ca2991a50522ba0f05ad0956cc52b0f6922957
3
+ size 184220842
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/3_trainer_state.json ADDED
@@ -0,0 +1,392 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 1.0,
5
+ "eval_steps": 500,
6
+ "global_step": 100,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.02,
13
+ "grad_norm": 11.693649291992188,
14
+ "learning_rate": 2e-05,
15
+ "loss": 1.0507,
16
+ "step": 2
17
+ },
18
+ {
19
+ "epoch": 0.04,
20
+ "grad_norm": 2.9311091899871826,
21
+ "learning_rate": 2e-05,
22
+ "loss": 0.4164,
23
+ "step": 4
24
+ },
25
+ {
26
+ "epoch": 0.06,
27
+ "grad_norm": 8.45428466796875,
28
+ "learning_rate": 2e-05,
29
+ "loss": 1.1758,
30
+ "step": 6
31
+ },
32
+ {
33
+ "epoch": 0.08,
34
+ "grad_norm": 14.28290843963623,
35
+ "learning_rate": 2e-05,
36
+ "loss": 1.3931,
37
+ "step": 8
38
+ },
39
+ {
40
+ "epoch": 0.1,
41
+ "grad_norm": 15.484107971191406,
42
+ "learning_rate": 2e-05,
43
+ "loss": 1.6668,
44
+ "step": 10
45
+ },
46
+ {
47
+ "epoch": 0.12,
48
+ "grad_norm": 9.924738883972168,
49
+ "learning_rate": 2e-05,
50
+ "loss": 1.5835,
51
+ "step": 12
52
+ },
53
+ {
54
+ "epoch": 0.14,
55
+ "grad_norm": 8.884476661682129,
56
+ "learning_rate": 2e-05,
57
+ "loss": 1.1145,
58
+ "step": 14
59
+ },
60
+ {
61
+ "epoch": 0.16,
62
+ "grad_norm": 9.459785461425781,
63
+ "learning_rate": 2e-05,
64
+ "loss": 0.9232,
65
+ "step": 16
66
+ },
67
+ {
68
+ "epoch": 0.18,
69
+ "grad_norm": 9.546226501464844,
70
+ "learning_rate": 2e-05,
71
+ "loss": 0.5492,
72
+ "step": 18
73
+ },
74
+ {
75
+ "epoch": 0.2,
76
+ "grad_norm": 2.949314594268799,
77
+ "learning_rate": 2e-05,
78
+ "loss": 0.744,
79
+ "step": 20
80
+ },
81
+ {
82
+ "epoch": 0.22,
83
+ "grad_norm": 5.894227027893066,
84
+ "learning_rate": 2e-05,
85
+ "loss": 0.6446,
86
+ "step": 22
87
+ },
88
+ {
89
+ "epoch": 0.24,
90
+ "grad_norm": 5.924652099609375,
91
+ "learning_rate": 2e-05,
92
+ "loss": 0.8909,
93
+ "step": 24
94
+ },
95
+ {
96
+ "epoch": 0.26,
97
+ "grad_norm": 2.6994128227233887,
98
+ "learning_rate": 2e-05,
99
+ "loss": 0.4439,
100
+ "step": 26
101
+ },
102
+ {
103
+ "epoch": 0.28,
104
+ "grad_norm": 8.177955627441406,
105
+ "learning_rate": 2e-05,
106
+ "loss": 0.4963,
107
+ "step": 28
108
+ },
109
+ {
110
+ "epoch": 0.3,
111
+ "grad_norm": 9.719780921936035,
112
+ "learning_rate": 2e-05,
113
+ "loss": 0.8525,
114
+ "step": 30
115
+ },
116
+ {
117
+ "epoch": 0.32,
118
+ "grad_norm": 22.681800842285156,
119
+ "learning_rate": 2e-05,
120
+ "loss": 1.8416,
121
+ "step": 32
122
+ },
123
+ {
124
+ "epoch": 0.34,
125
+ "grad_norm": 3.0644638538360596,
126
+ "learning_rate": 2e-05,
127
+ "loss": 2.1517,
128
+ "step": 34
129
+ },
130
+ {
131
+ "epoch": 0.36,
132
+ "grad_norm": 14.563640594482422,
133
+ "learning_rate": 2e-05,
134
+ "loss": 2.4433,
135
+ "step": 36
136
+ },
137
+ {
138
+ "epoch": 0.38,
139
+ "grad_norm": 7.03754186630249,
140
+ "learning_rate": 2e-05,
141
+ "loss": 1.3491,
142
+ "step": 38
143
+ },
144
+ {
145
+ "epoch": 0.4,
146
+ "grad_norm": 6.36046838760376,
147
+ "learning_rate": 2e-05,
148
+ "loss": 1.1212,
149
+ "step": 40
150
+ },
151
+ {
152
+ "epoch": 0.42,
153
+ "grad_norm": 18.849185943603516,
154
+ "learning_rate": 2e-05,
155
+ "loss": 1.6678,
156
+ "step": 42
157
+ },
158
+ {
159
+ "epoch": 0.44,
160
+ "grad_norm": 14.333285331726074,
161
+ "learning_rate": 2e-05,
162
+ "loss": 1.6081,
163
+ "step": 44
164
+ },
165
+ {
166
+ "epoch": 0.46,
167
+ "grad_norm": 6.0595269203186035,
168
+ "learning_rate": 2e-05,
169
+ "loss": 0.6205,
170
+ "step": 46
171
+ },
172
+ {
173
+ "epoch": 0.48,
174
+ "grad_norm": 4.74971342086792,
175
+ "learning_rate": 2e-05,
176
+ "loss": 0.4097,
177
+ "step": 48
178
+ },
179
+ {
180
+ "epoch": 0.5,
181
+ "grad_norm": 7.75990104675293,
182
+ "learning_rate": 2e-05,
183
+ "loss": 1.074,
184
+ "step": 50
185
+ },
186
+ {
187
+ "epoch": 0.52,
188
+ "grad_norm": 9.010882377624512,
189
+ "learning_rate": 2e-05,
190
+ "loss": 2.1047,
191
+ "step": 52
192
+ },
193
+ {
194
+ "epoch": 0.54,
195
+ "grad_norm": 1.1840580701828003,
196
+ "learning_rate": 2e-05,
197
+ "loss": 0.8681,
198
+ "step": 54
199
+ },
200
+ {
201
+ "epoch": 0.56,
202
+ "grad_norm": 2.5251049995422363,
203
+ "learning_rate": 2e-05,
204
+ "loss": 0.7495,
205
+ "step": 56
206
+ },
207
+ {
208
+ "epoch": 0.58,
209
+ "grad_norm": 6.284056186676025,
210
+ "learning_rate": 2e-05,
211
+ "loss": 0.8752,
212
+ "step": 58
213
+ },
214
+ {
215
+ "epoch": 0.6,
216
+ "grad_norm": 8.168758392333984,
217
+ "learning_rate": 2e-05,
218
+ "loss": 1.3925,
219
+ "step": 60
220
+ },
221
+ {
222
+ "epoch": 0.62,
223
+ "grad_norm": 6.6547770500183105,
224
+ "learning_rate": 2e-05,
225
+ "loss": 0.7414,
226
+ "step": 62
227
+ },
228
+ {
229
+ "epoch": 0.64,
230
+ "grad_norm": 7.461399078369141,
231
+ "learning_rate": 2e-05,
232
+ "loss": 0.8258,
233
+ "step": 64
234
+ },
235
+ {
236
+ "epoch": 0.66,
237
+ "grad_norm": 6.963188648223877,
238
+ "learning_rate": 2e-05,
239
+ "loss": 1.218,
240
+ "step": 66
241
+ },
242
+ {
243
+ "epoch": 0.68,
244
+ "grad_norm": 5.871964454650879,
245
+ "learning_rate": 2e-05,
246
+ "loss": 1.1124,
247
+ "step": 68
248
+ },
249
+ {
250
+ "epoch": 0.7,
251
+ "grad_norm": 7.110522747039795,
252
+ "learning_rate": 2e-05,
253
+ "loss": 1.2579,
254
+ "step": 70
255
+ },
256
+ {
257
+ "epoch": 0.72,
258
+ "grad_norm": 8.83693790435791,
259
+ "learning_rate": 2e-05,
260
+ "loss": 2.2598,
261
+ "step": 72
262
+ },
263
+ {
264
+ "epoch": 0.74,
265
+ "grad_norm": 7.1040520668029785,
266
+ "learning_rate": 2e-05,
267
+ "loss": 1.2506,
268
+ "step": 74
269
+ },
270
+ {
271
+ "epoch": 0.76,
272
+ "grad_norm": 17.06218719482422,
273
+ "learning_rate": 2e-05,
274
+ "loss": 1.4432,
275
+ "step": 76
276
+ },
277
+ {
278
+ "epoch": 0.78,
279
+ "grad_norm": 7.473016262054443,
280
+ "learning_rate": 2e-05,
281
+ "loss": 0.784,
282
+ "step": 78
283
+ },
284
+ {
285
+ "epoch": 0.8,
286
+ "grad_norm": 8.905854225158691,
287
+ "learning_rate": 2e-05,
288
+ "loss": 0.8411,
289
+ "step": 80
290
+ },
291
+ {
292
+ "epoch": 0.82,
293
+ "grad_norm": 4.500631332397461,
294
+ "learning_rate": 2e-05,
295
+ "loss": 0.2397,
296
+ "step": 82
297
+ },
298
+ {
299
+ "epoch": 0.84,
300
+ "grad_norm": 4.327226638793945,
301
+ "learning_rate": 2e-05,
302
+ "loss": 0.4464,
303
+ "step": 84
304
+ },
305
+ {
306
+ "epoch": 0.86,
307
+ "grad_norm": 4.575321674346924,
308
+ "learning_rate": 2e-05,
309
+ "loss": 1.1319,
310
+ "step": 86
311
+ },
312
+ {
313
+ "epoch": 0.88,
314
+ "grad_norm": 4.218846321105957,
315
+ "learning_rate": 2e-05,
316
+ "loss": 0.8927,
317
+ "step": 88
318
+ },
319
+ {
320
+ "epoch": 0.9,
321
+ "grad_norm": 7.213256359100342,
322
+ "learning_rate": 2e-05,
323
+ "loss": 0.9017,
324
+ "step": 90
325
+ },
326
+ {
327
+ "epoch": 0.92,
328
+ "grad_norm": 10.291433334350586,
329
+ "learning_rate": 2e-05,
330
+ "loss": 0.6296,
331
+ "step": 92
332
+ },
333
+ {
334
+ "epoch": 0.94,
335
+ "grad_norm": 6.705934524536133,
336
+ "learning_rate": 2e-05,
337
+ "loss": 0.6399,
338
+ "step": 94
339
+ },
340
+ {
341
+ "epoch": 0.96,
342
+ "grad_norm": 3.8862531185150146,
343
+ "learning_rate": 2e-05,
344
+ "loss": 0.3221,
345
+ "step": 96
346
+ },
347
+ {
348
+ "epoch": 0.98,
349
+ "grad_norm": 9.070793151855469,
350
+ "learning_rate": 2e-05,
351
+ "loss": 1.1441,
352
+ "step": 98
353
+ },
354
+ {
355
+ "epoch": 1.0,
356
+ "grad_norm": 11.29675006866455,
357
+ "learning_rate": 2e-05,
358
+ "loss": 1.2231,
359
+ "step": 100
360
+ },
361
+ {
362
+ "epoch": 1.0,
363
+ "step": 100,
364
+ "total_flos": 2097655350034432.0,
365
+ "train_loss": 1.0705613040924071,
366
+ "train_runtime": 102.5888,
367
+ "train_samples_per_second": 3.899,
368
+ "train_steps_per_second": 0.975
369
+ }
370
+ ],
371
+ "logging_steps": 2,
372
+ "max_steps": 100,
373
+ "num_input_tokens_seen": 0,
374
+ "num_train_epochs": 1,
375
+ "save_steps": 500,
376
+ "stateful_callbacks": {
377
+ "TrainerControl": {
378
+ "args": {
379
+ "should_epoch_stop": false,
380
+ "should_evaluate": false,
381
+ "should_log": false,
382
+ "should_save": false,
383
+ "should_training_stop": false
384
+ },
385
+ "attributes": {}
386
+ }
387
+ },
388
+ "total_flos": 2097655350034432.0,
389
+ "train_batch_size": 1,
390
+ "trial_name": null,
391
+ "trial_params": null
392
+ }
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round10.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:63c020f0e9a8f0c812148e6b98c897be1fbf16e0e6bdf9c1cb855bd2fb184967
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a576f6efc9cbd3bb8a501153bc3e085ec89a15902ad8cfe10bc4b974f03e4fcd
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round15.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:5db05f3d3601799e1fc60bea63f67d5557eee27fabad09326273c4391caa1f03
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round17.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:7b1615aeafa8813c4d2376f2ae7a14b957139e2e63d429b39c3fb322e8bde63a
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:6ba947d6751d0af4d240a2ac8b103cec9438b64bef846a3c69fd34bf8554f5df
3
+ size 395786922
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round20.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:56fa94a36ca7fe11a494c594eb9b640dd6ee9b47daeee27e2fee4d62575c6f43
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac555e618f6ec228443d7e19fcd469b8ef8e5bea20f651ef5cde9cdf85221ad9
3
+ size 395786922
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_client_model_round7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9e7a27691ec0a074141418aadc3a9c16e25bbeb6a18c82bc76883e6acb92cb24
3
+ size 395786922
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/4_trainer_state.json ADDED
@@ -0,0 +1,392 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_metric": null,
3
+ "best_model_checkpoint": null,
4
+ "epoch": 1.0,
5
+ "eval_steps": 500,
6
+ "global_step": 100,
7
+ "is_hyper_param_search": false,
8
+ "is_local_process_zero": true,
9
+ "is_world_process_zero": true,
10
+ "log_history": [
11
+ {
12
+ "epoch": 0.02,
13
+ "grad_norm": 2.637403726577759,
14
+ "learning_rate": 2e-05,
15
+ "loss": 0.5614,
16
+ "step": 2
17
+ },
18
+ {
19
+ "epoch": 0.04,
20
+ "grad_norm": 4.73873233795166,
21
+ "learning_rate": 2e-05,
22
+ "loss": 0.9697,
23
+ "step": 4
24
+ },
25
+ {
26
+ "epoch": 0.06,
27
+ "grad_norm": 1.6665256023406982,
28
+ "learning_rate": 2e-05,
29
+ "loss": 0.4683,
30
+ "step": 6
31
+ },
32
+ {
33
+ "epoch": 0.08,
34
+ "grad_norm": 3.0230000019073486,
35
+ "learning_rate": 2e-05,
36
+ "loss": 0.6475,
37
+ "step": 8
38
+ },
39
+ {
40
+ "epoch": 0.1,
41
+ "grad_norm": 2.887681245803833,
42
+ "learning_rate": 2e-05,
43
+ "loss": 0.5517,
44
+ "step": 10
45
+ },
46
+ {
47
+ "epoch": 0.12,
48
+ "grad_norm": 3.429380416870117,
49
+ "learning_rate": 2e-05,
50
+ "loss": 0.2907,
51
+ "step": 12
52
+ },
53
+ {
54
+ "epoch": 0.14,
55
+ "grad_norm": 4.610021114349365,
56
+ "learning_rate": 2e-05,
57
+ "loss": 0.6544,
58
+ "step": 14
59
+ },
60
+ {
61
+ "epoch": 0.16,
62
+ "grad_norm": 8.516341209411621,
63
+ "learning_rate": 2e-05,
64
+ "loss": 1.9958,
65
+ "step": 16
66
+ },
67
+ {
68
+ "epoch": 0.18,
69
+ "grad_norm": 3.5461740493774414,
70
+ "learning_rate": 2e-05,
71
+ "loss": 0.3917,
72
+ "step": 18
73
+ },
74
+ {
75
+ "epoch": 0.2,
76
+ "grad_norm": 5.2099456787109375,
77
+ "learning_rate": 2e-05,
78
+ "loss": 0.4288,
79
+ "step": 20
80
+ },
81
+ {
82
+ "epoch": 0.22,
83
+ "grad_norm": 4.585675239562988,
84
+ "learning_rate": 2e-05,
85
+ "loss": 1.801,
86
+ "step": 22
87
+ },
88
+ {
89
+ "epoch": 0.24,
90
+ "grad_norm": 2.1597061157226562,
91
+ "learning_rate": 2e-05,
92
+ "loss": 1.0035,
93
+ "step": 24
94
+ },
95
+ {
96
+ "epoch": 0.26,
97
+ "grad_norm": 3.1741673946380615,
98
+ "learning_rate": 2e-05,
99
+ "loss": 0.4851,
100
+ "step": 26
101
+ },
102
+ {
103
+ "epoch": 0.28,
104
+ "grad_norm": 5.476759433746338,
105
+ "learning_rate": 2e-05,
106
+ "loss": 1.1165,
107
+ "step": 28
108
+ },
109
+ {
110
+ "epoch": 0.3,
111
+ "grad_norm": 4.2731804847717285,
112
+ "learning_rate": 2e-05,
113
+ "loss": 0.9401,
114
+ "step": 30
115
+ },
116
+ {
117
+ "epoch": 0.32,
118
+ "grad_norm": 6.725061893463135,
119
+ "learning_rate": 2e-05,
120
+ "loss": 0.7967,
121
+ "step": 32
122
+ },
123
+ {
124
+ "epoch": 0.34,
125
+ "grad_norm": 2.4326884746551514,
126
+ "learning_rate": 2e-05,
127
+ "loss": 0.4525,
128
+ "step": 34
129
+ },
130
+ {
131
+ "epoch": 0.36,
132
+ "grad_norm": 4.693164348602295,
133
+ "learning_rate": 2e-05,
134
+ "loss": 1.2137,
135
+ "step": 36
136
+ },
137
+ {
138
+ "epoch": 0.38,
139
+ "grad_norm": 2.8988044261932373,
140
+ "learning_rate": 2e-05,
141
+ "loss": 0.5093,
142
+ "step": 38
143
+ },
144
+ {
145
+ "epoch": 0.4,
146
+ "grad_norm": 6.970594882965088,
147
+ "learning_rate": 2e-05,
148
+ "loss": 1.0991,
149
+ "step": 40
150
+ },
151
+ {
152
+ "epoch": 0.42,
153
+ "grad_norm": 3.4238510131835938,
154
+ "learning_rate": 2e-05,
155
+ "loss": 0.8502,
156
+ "step": 42
157
+ },
158
+ {
159
+ "epoch": 0.44,
160
+ "grad_norm": 3.9824445247650146,
161
+ "learning_rate": 2e-05,
162
+ "loss": 0.6589,
163
+ "step": 44
164
+ },
165
+ {
166
+ "epoch": 0.46,
167
+ "grad_norm": 4.8126678466796875,
168
+ "learning_rate": 2e-05,
169
+ "loss": 0.8948,
170
+ "step": 46
171
+ },
172
+ {
173
+ "epoch": 0.48,
174
+ "grad_norm": 5.100122451782227,
175
+ "learning_rate": 2e-05,
176
+ "loss": 0.8117,
177
+ "step": 48
178
+ },
179
+ {
180
+ "epoch": 0.5,
181
+ "grad_norm": 2.1359190940856934,
182
+ "learning_rate": 2e-05,
183
+ "loss": 0.4148,
184
+ "step": 50
185
+ },
186
+ {
187
+ "epoch": 0.52,
188
+ "grad_norm": 5.385488033294678,
189
+ "learning_rate": 2e-05,
190
+ "loss": 1.6419,
191
+ "step": 52
192
+ },
193
+ {
194
+ "epoch": 0.54,
195
+ "grad_norm": 4.634507656097412,
196
+ "learning_rate": 2e-05,
197
+ "loss": 0.8469,
198
+ "step": 54
199
+ },
200
+ {
201
+ "epoch": 0.56,
202
+ "grad_norm": 2.4589309692382812,
203
+ "learning_rate": 2e-05,
204
+ "loss": 0.2448,
205
+ "step": 56
206
+ },
207
+ {
208
+ "epoch": 0.58,
209
+ "grad_norm": 4.612209796905518,
210
+ "learning_rate": 2e-05,
211
+ "loss": 0.836,
212
+ "step": 58
213
+ },
214
+ {
215
+ "epoch": 0.6,
216
+ "grad_norm": 3.2502317428588867,
217
+ "learning_rate": 2e-05,
218
+ "loss": 0.8215,
219
+ "step": 60
220
+ },
221
+ {
222
+ "epoch": 0.62,
223
+ "grad_norm": 3.3926491737365723,
224
+ "learning_rate": 2e-05,
225
+ "loss": 0.6394,
226
+ "step": 62
227
+ },
228
+ {
229
+ "epoch": 0.64,
230
+ "grad_norm": 3.6580755710601807,
231
+ "learning_rate": 2e-05,
232
+ "loss": 1.1433,
233
+ "step": 64
234
+ },
235
+ {
236
+ "epoch": 0.66,
237
+ "grad_norm": 7.191780090332031,
238
+ "learning_rate": 2e-05,
239
+ "loss": 1.0154,
240
+ "step": 66
241
+ },
242
+ {
243
+ "epoch": 0.68,
244
+ "grad_norm": 2.0875039100646973,
245
+ "learning_rate": 2e-05,
246
+ "loss": 0.6591,
247
+ "step": 68
248
+ },
249
+ {
250
+ "epoch": 0.7,
251
+ "grad_norm": 7.09459114074707,
252
+ "learning_rate": 2e-05,
253
+ "loss": 0.8218,
254
+ "step": 70
255
+ },
256
+ {
257
+ "epoch": 0.72,
258
+ "grad_norm": 3.325845241546631,
259
+ "learning_rate": 2e-05,
260
+ "loss": 0.6073,
261
+ "step": 72
262
+ },
263
+ {
264
+ "epoch": 0.74,
265
+ "grad_norm": 1.3921314477920532,
266
+ "learning_rate": 2e-05,
267
+ "loss": 0.5865,
268
+ "step": 74
269
+ },
270
+ {
271
+ "epoch": 0.76,
272
+ "grad_norm": 3.6516613960266113,
273
+ "learning_rate": 2e-05,
274
+ "loss": 1.4063,
275
+ "step": 76
276
+ },
277
+ {
278
+ "epoch": 0.78,
279
+ "grad_norm": 4.696588039398193,
280
+ "learning_rate": 2e-05,
281
+ "loss": 0.7179,
282
+ "step": 78
283
+ },
284
+ {
285
+ "epoch": 0.8,
286
+ "grad_norm": 3.545356512069702,
287
+ "learning_rate": 2e-05,
288
+ "loss": 0.4085,
289
+ "step": 80
290
+ },
291
+ {
292
+ "epoch": 0.82,
293
+ "grad_norm": 3.8088479042053223,
294
+ "learning_rate": 2e-05,
295
+ "loss": 0.5266,
296
+ "step": 82
297
+ },
298
+ {
299
+ "epoch": 0.84,
300
+ "grad_norm": 5.2242326736450195,
301
+ "learning_rate": 2e-05,
302
+ "loss": 0.6016,
303
+ "step": 84
304
+ },
305
+ {
306
+ "epoch": 0.86,
307
+ "grad_norm": 2.499507427215576,
308
+ "learning_rate": 2e-05,
309
+ "loss": 0.3916,
310
+ "step": 86
311
+ },
312
+ {
313
+ "epoch": 0.88,
314
+ "grad_norm": 3.5232198238372803,
315
+ "learning_rate": 2e-05,
316
+ "loss": 0.5096,
317
+ "step": 88
318
+ },
319
+ {
320
+ "epoch": 0.9,
321
+ "grad_norm": 4.562027931213379,
322
+ "learning_rate": 2e-05,
323
+ "loss": 0.7677,
324
+ "step": 90
325
+ },
326
+ {
327
+ "epoch": 0.92,
328
+ "grad_norm": 6.415626525878906,
329
+ "learning_rate": 2e-05,
330
+ "loss": 0.7187,
331
+ "step": 92
332
+ },
333
+ {
334
+ "epoch": 0.94,
335
+ "grad_norm": 5.6967644691467285,
336
+ "learning_rate": 2e-05,
337
+ "loss": 0.7985,
338
+ "step": 94
339
+ },
340
+ {
341
+ "epoch": 0.96,
342
+ "grad_norm": 3.2610225677490234,
343
+ "learning_rate": 2e-05,
344
+ "loss": 0.6187,
345
+ "step": 96
346
+ },
347
+ {
348
+ "epoch": 0.98,
349
+ "grad_norm": 3.403749942779541,
350
+ "learning_rate": 2e-05,
351
+ "loss": 0.2896,
352
+ "step": 98
353
+ },
354
+ {
355
+ "epoch": 1.0,
356
+ "grad_norm": 11.633613586425781,
357
+ "learning_rate": 2e-05,
358
+ "loss": 1.1527,
359
+ "step": 100
360
+ },
361
+ {
362
+ "epoch": 1.0,
363
+ "step": 100,
364
+ "total_flos": 5694661670731776.0,
365
+ "train_loss": 0.7755919742584229,
366
+ "train_runtime": 170.0795,
367
+ "train_samples_per_second": 2.352,
368
+ "train_steps_per_second": 0.588
369
+ }
370
+ ],
371
+ "logging_steps": 2,
372
+ "max_steps": 100,
373
+ "num_input_tokens_seen": 0,
374
+ "num_train_epochs": 1,
375
+ "save_steps": 500,
376
+ "stateful_callbacks": {
377
+ "TrainerControl": {
378
+ "args": {
379
+ "should_epoch_stop": false,
380
+ "should_evaluate": false,
381
+ "should_log": false,
382
+ "should_save": false,
383
+ "should_training_stop": false
384
+ },
385
+ "attributes": {}
386
+ }
387
+ },
388
+ "total_flos": 5694661670731776.0,
389
+ "train_batch_size": 1,
390
+ "trial_name": null,
391
+ "trial_params": null
392
+ }
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round10.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:75fce352d5052d2b03312c457b9d07a3c2045a3c1d82fa169cde815ffac140a0
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round12.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:51efd3d761dbc6a85da1d36fd13afc6913e8f35c7c18b9825c9a03ebd8725119
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round15.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:04acc4730d61aa5cab2b5961a914144dea9077c80a98ea800ef7b634fb202740
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round17.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c21f2c0bf74b5e28d521eff1cae89902767d9acad0a560a77818c53885c72763
3
+ size 395787774
client_states_fedMulti05pqfullfreeze_sft_pca_bs4_saveoptim_lr2e-5_5e-5_sc315_4tasks_5rounds_fixitr100_T0125_decay099/5_client_model_round2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cf27c9f6c67ee9af8765ac0ea5533f11f5d1e729b6e3d53daf4b640cb3853670
3
+ size 395786922