heavyhelium commited on
Commit
acc5d25
·
verified ·
1 Parent(s): fa7227f

Training in progress, epoch 1, checkpoint

Browse files
checkpoint-93/model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:b8382937490df6137c1fd1d38ba0f2b1b42eb923ae848f769240e5dc21ea8533
3
  size 368871908
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:c0774576e3bd0e6798212b0c9287b0a20f8f8b3267d38eeacc15254043250403
3
  size 368871908
checkpoint-93/optimizer.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:05cdfe2df56b20774c596d833eca8b61230989f2f5c39397f44183d499bafc23
3
  size 737866955
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f277c9cfafc7909d4a67acc39d36ad1066fc77f52e9cafdecdc4f5a7471d4b6d
3
  size 737866955
checkpoint-93/scheduler.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:95e5b942d62434299e26ae77a7817d0da66f975a4b4278276a96ab34fdabade3
3
  size 1465
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1509f821eeec3d1883e4c8feb1c9a82904e7b01ad278fdb204cbc4aebc86cbb0
3
  size 1465
checkpoint-93/trainer_state.json CHANGED
@@ -11,76 +11,76 @@
11
  "log_history": [
12
  {
13
  "epoch": 0.10752688172043011,
14
- "grad_norm": 4.2265625,
15
- "learning_rate": 3.2142857142857144e-05,
16
- "loss": 0.7183075428009034,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.21505376344086022,
21
- "grad_norm": 2.408203125,
22
- "learning_rate": 6.785714285714286e-05,
23
- "loss": 0.6934913635253906,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.3225806451612903,
28
- "grad_norm": NaN,
29
- "learning_rate": 9.960159362549801e-05,
30
- "loss": 0.0,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.43010752688172044,
35
- "grad_norm": NaN,
36
- "learning_rate": 9.56175298804781e-05,
37
- "loss": 0.0,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.5376344086021505,
42
- "grad_norm": NaN,
43
- "learning_rate": 9.163346613545816e-05,
44
- "loss": 0.0,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.6451612903225806,
49
- "grad_norm": NaN,
50
- "learning_rate": 8.764940239043824e-05,
51
- "loss": 0.0,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.7526881720430108,
56
- "grad_norm": NaN,
57
- "learning_rate": 8.366533864541834e-05,
58
- "loss": 0.0,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.8602150537634409,
63
- "grad_norm": NaN,
64
- "learning_rate": 7.968127490039841e-05,
65
- "loss": 0.0,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.967741935483871,
70
- "grad_norm": NaN,
71
- "learning_rate": 7.569721115537849e-05,
72
- "loss": 0.0,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 1.0,
77
  "eval_accuracy": 0.5,
78
- "eval_fallacy_f1": 0.0,
79
- "eval_loss": NaN,
80
  "eval_macro_f1": 0.3333333333333333,
81
- "eval_runtime": 0.7269,
82
- "eval_samples_per_second": 275.146,
83
- "eval_steps_per_second": 9.63,
84
  "step": 93
85
  }
86
  ],
 
11
  "log_history": [
12
  {
13
  "epoch": 0.10752688172043011,
14
+ "grad_norm": 3.478515625,
15
+ "learning_rate": 6.4285714285714295e-06,
16
+ "loss": 0.7033989906311036,
17
  "step": 10
18
  },
19
  {
20
  "epoch": 0.21505376344086022,
21
+ "grad_norm": 2.998046875,
22
+ "learning_rate": 1.3571428571428574e-05,
23
+ "loss": 0.7003115653991699,
24
  "step": 20
25
  },
26
  {
27
  "epoch": 0.3225806451612903,
28
+ "grad_norm": 1.7021484375,
29
+ "learning_rate": 1.9920318725099602e-05,
30
+ "loss": 0.712247085571289,
31
  "step": 30
32
  },
33
  {
34
  "epoch": 0.43010752688172044,
35
+ "grad_norm": 1.376953125,
36
+ "learning_rate": 1.912350597609562e-05,
37
+ "loss": 0.7091613292694092,
38
  "step": 40
39
  },
40
  {
41
  "epoch": 0.5376344086021505,
42
+ "grad_norm": 2.8359375,
43
+ "learning_rate": 1.8326693227091633e-05,
44
+ "loss": 0.7023275852203369,
45
  "step": 50
46
  },
47
  {
48
  "epoch": 0.6451612903225806,
49
+ "grad_norm": 3.02734375,
50
+ "learning_rate": 1.752988047808765e-05,
51
+ "loss": 0.6804835796356201,
52
  "step": 60
53
  },
54
  {
55
  "epoch": 0.7526881720430108,
56
+ "grad_norm": 2.419921875,
57
+ "learning_rate": 1.6733067729083667e-05,
58
+ "loss": 0.711079454421997,
59
  "step": 70
60
  },
61
  {
62
  "epoch": 0.8602150537634409,
63
+ "grad_norm": 3.498046875,
64
+ "learning_rate": 1.593625498007968e-05,
65
+ "loss": 0.7339372158050537,
66
  "step": 80
67
  },
68
  {
69
  "epoch": 0.967741935483871,
70
+ "grad_norm": 1.2763671875,
71
+ "learning_rate": 1.5139442231075698e-05,
72
+ "loss": 0.6859766960144043,
73
  "step": 90
74
  },
75
  {
76
  "epoch": 1.0,
77
  "eval_accuracy": 0.5,
78
+ "eval_fallacy_f1": 0.6666666666666666,
79
+ "eval_loss": 0.7136235237121582,
80
  "eval_macro_f1": 0.3333333333333333,
81
+ "eval_runtime": 0.8023,
82
+ "eval_samples_per_second": 249.276,
83
+ "eval_steps_per_second": 8.725,
84
  "step": 93
85
  }
86
  ],
checkpoint-93/training_args.bin CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:8e17d511c1800a8ae9a6d0680e36858be596dfb09ca1b8074fe7ce9f2d264ecf
3
  size 5393
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1ad3444cd2e07351dbe64a248b05135aa3a21df313623d7cf200ed75ec9c73fb
3
  size 5393