HoangVuSnape commited on
Commit
d29e089
·
verified ·
1 Parent(s): ae2944b

Training in progress, step 150, checkpoint

Browse files
checkpoint-150/adapter_model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:613380280d3cd56dd958ffa13b63bc2bba19e9a4bbab89d8ace7f962baaf1865
3
  size 165012392
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:8deb0060d7358aeabd1230137b33365c199273f1adea1fb63f5896477c8c3c65
3
  size 165012392
checkpoint-150/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:e41a29e47e1a17cd3a5a368ad2a24e159a4f38da0f71f5801290242d6ebd5521
3
  size 84690389
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f6e5566d2fb0d78e03b6c33471dcf154fb8928ec5659362421bf60549d15e1f0
3
  size 84690389
checkpoint-150/trainer_state.json CHANGED
@@ -11,115 +11,115 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.022308979364194088,
14
- "grad_norm": 4.757451057434082,
15
  "learning_rate": 0.00019999446787180336,
16
- "loss": 3.6770606994628907,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.044617958728388175,
21
- "grad_norm": 1.3747667074203491,
22
  "learning_rate": 0.00019989613595281384,
23
- "loss": 1.009000873565674,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.06692693809258227,
28
- "grad_norm": 1.0401424169540405,
29
  "learning_rate": 0.00019967500698688952,
30
- "loss": 0.6844106197357178,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.08923591745677635,
35
- "grad_norm": 0.8778170347213745,
36
  "learning_rate": 0.0001993313527961959,
37
- "loss": 0.5973338603973388,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.11154489682097044,
42
- "grad_norm": 1.0252398252487183,
43
  "learning_rate": 0.00019886559581668999,
44
- "loss": 0.5388701915740967,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.13385387618516453,
49
- "grad_norm": 0.9101352095603943,
50
  "learning_rate": 0.00019827830857884173,
51
- "loss": 0.44642319679260256,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.1561628555493586,
56
- "grad_norm": 0.8221309185028076,
57
  "learning_rate": 0.00019757021300385286,
58
- "loss": 0.4484751224517822,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.1784718349135527,
63
- "grad_norm": 0.5932356119155884,
64
  "learning_rate": 0.00019674217951623707,
65
- "loss": 0.40591115951538087,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.2007808142777468,
70
- "grad_norm": 0.7747540473937988,
71
  "learning_rate": 0.00019579522597385315,
72
- "loss": 0.38509316444396974,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.22308979364194087,
77
- "grad_norm": 0.6519433259963989,
78
  "learning_rate": 0.00019473051641670606,
79
- "loss": 0.38065552711486816,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.22308979364194087,
84
- "eval_loss": 8.738031387329102,
85
- "eval_runtime": 244.6378,
86
- "eval_samples_per_second": 1.844,
87
- "eval_steps_per_second": 1.844,
88
  "step": 100
89
  },
90
  {
91
  "epoch": 0.24539877300613497,
92
- "grad_norm": 0.5722377896308899,
93
  "learning_rate": 0.00019354935963605393,
94
- "loss": 0.3379225730895996,
95
  "step": 110
96
  },
97
  {
98
  "epoch": 0.26770775237032907,
99
- "grad_norm": 0.6395815014839172,
100
  "learning_rate": 0.00019225320756558023,
101
- "loss": 0.3527653932571411,
102
  "step": 120
103
  },
104
  {
105
  "epoch": 0.29001673173452314,
106
- "grad_norm": 0.5309840440750122,
107
  "learning_rate": 0.0001908436534966081,
108
- "loss": 0.32701103687286376,
109
  "step": 130
110
  },
111
  {
112
  "epoch": 0.3123257110987172,
113
- "grad_norm": 1.1165753602981567,
114
  "learning_rate": 0.00018932243011955154,
115
- "loss": 0.42706594467163084,
116
  "step": 140
117
  },
118
  {
119
  "epoch": 0.33463469046291133,
120
- "grad_norm": 0.5633851885795593,
121
  "learning_rate": 0.00018769140739401062,
122
- "loss": 0.2921785354614258,
123
  "step": 150
124
  }
125
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.022308979364194088,
14
+ "grad_norm": 4.709710597991943,
15
  "learning_rate": 0.00019999446787180336,
16
+ "loss": 3.676974868774414,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.044617958728388175,
21
+ "grad_norm": 1.4567488431930542,
22
  "learning_rate": 0.00019989613595281384,
23
+ "loss": 0.9999975204467774,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.06692693809258227,
28
+ "grad_norm": 1.1962672472000122,
29
  "learning_rate": 0.00019967500698688952,
30
+ "loss": 0.6824670314788819,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.08923591745677635,
35
+ "grad_norm": 0.8628137707710266,
36
  "learning_rate": 0.0001993313527961959,
37
+ "loss": 0.5993218421936035,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.11154489682097044,
42
+ "grad_norm": 1.0308934450149536,
43
  "learning_rate": 0.00019886559581668999,
44
+ "loss": 0.542162561416626,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.13385387618516453,
49
+ "grad_norm": 0.9261613488197327,
50
  "learning_rate": 0.00019827830857884173,
51
+ "loss": 0.44734554290771483,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.1561628555493586,
56
+ "grad_norm": 0.9318423867225647,
57
  "learning_rate": 0.00019757021300385286,
58
+ "loss": 0.44412760734558104,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.1784718349135527,
63
+ "grad_norm": 0.5784896612167358,
64
  "learning_rate": 0.00019674217951623707,
65
+ "loss": 0.40378632545471194,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.2007808142777468,
70
+ "grad_norm": 0.7836484909057617,
71
  "learning_rate": 0.00019579522597385315,
72
+ "loss": 0.3856403350830078,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 0.22308979364194087,
77
+ "grad_norm": 0.6503493189811707,
78
  "learning_rate": 0.00019473051641670606,
79
+ "loss": 0.3800614356994629,
80
  "step": 100
81
  },
82
  {
83
  "epoch": 0.22308979364194087,
84
+ "eval_loss": 8.158984184265137,
85
+ "eval_runtime": 248.821,
86
+ "eval_samples_per_second": 1.813,
87
+ "eval_steps_per_second": 1.813,
88
  "step": 100
89
  },
90
  {
91
  "epoch": 0.24539877300613497,
92
+ "grad_norm": 0.582433819770813,
93
  "learning_rate": 0.00019354935963605393,
94
+ "loss": 0.34090685844421387,
95
  "step": 110
96
  },
97
  {
98
  "epoch": 0.26770775237032907,
99
+ "grad_norm": 0.6219270825386047,
100
  "learning_rate": 0.00019225320756558023,
101
+ "loss": 0.3560009956359863,
102
  "step": 120
103
  },
104
  {
105
  "epoch": 0.29001673173452314,
106
+ "grad_norm": 0.5081164836883545,
107
  "learning_rate": 0.0001908436534966081,
108
+ "loss": 0.32209837436676025,
109
  "step": 130
110
  },
111
  {
112
  "epoch": 0.3123257110987172,
113
+ "grad_norm": 1.0672804117202759,
114
  "learning_rate": 0.00018932243011955154,
115
+ "loss": 0.4280549049377441,
116
  "step": 140
117
  },
118
  {
119
  "epoch": 0.33463469046291133,
120
+ "grad_norm": 0.5716775059700012,
121
  "learning_rate": 0.00018769140739401062,
122
+ "loss": 0.29103860855102537,
123
  "step": 150
124
  }
125
  ],