CodeIsAbstract commited on
Commit
e86afbb
·
verified ·
1 Parent(s): 25a88db

Training in progress, step 500, checkpoint

Browse files
last-checkpoint/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:80d69a236e68b6332db7d3cff5deaa06ec25ff0e5efa550677ce2f35df088680
3
+ size 386379
last-checkpoint/pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:ac6bc441453e9b4734e133be3c600607e5b3dee32b860836792368241f27642d
3
+ size 1540661735
last-checkpoint/rng_state_0.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:43badff6712b33504459f872ac58be74afdf5ffa731e863439dbed8d3f3b0658
3
+ size 14469
last-checkpoint/rng_state_1.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:12eac59d8227e48f0e6fc2526645da1f69a0138bb2904859d6e0d18216ba63d7
3
+ size 14469
last-checkpoint/rng_state_2.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:9c12b9a050992357585c15ce67795abc69719e584223ec4ccbae89612840d277
3
+ size 14469
last-checkpoint/rng_state_3.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:da83cd5eef961d44e4becc44e4089edf261063947ef5a08e30404cfbd611e2c3
3
+ size 14469
last-checkpoint/rng_state_4.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:bfd920f545c0711cea231c43d68bcba64df0cc8fd9e536bc95aea4ceced55b64
3
+ size 14469
last-checkpoint/rng_state_5.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e5e310e83ca23aa2c5186dec200b69809f8eb565d4f17f262d28f343cd531776
3
+ size 14469
last-checkpoint/rng_state_6.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:67aa5039be38b0a461bcbf305aaf4eb5ca5764f1670f75af2089e433b412f3af
3
+ size 14469
last-checkpoint/rng_state_7.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b2c15db3462043815a7b702c6ec5d03f93688f7392547928696d4f1b5bc14356
3
+ size 14469
last-checkpoint/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:e31938cdcda9330861d9a6c5fa22f1c1b3736037ca4758e7ec127110a6ee0d1a
3
+ size 1465
last-checkpoint/trainer_state.json ADDED
@@ -0,0 +1,122 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.16666666666666666,
6
+ "eval_steps": 250,
7
+ "global_step": 500,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": false,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.016666666666666666,
14
+ "grad_norm": 1325.952880859375,
15
+ "learning_rate": 0.00013066666666666668,
16
+ "loss": 1512.93171875,
17
+ "step": 50
18
+ },
19
+ {
20
+ "epoch": 0.03333333333333333,
21
+ "grad_norm": 929.898681640625,
22
+ "learning_rate": 0.000264,
23
+ "loss": 1464.5940625,
24
+ "step": 100
25
+ },
26
+ {
27
+ "epoch": 0.05,
28
+ "grad_norm": 747.4642333984375,
29
+ "learning_rate": 0.00039733333333333336,
30
+ "loss": 1415.48578125,
31
+ "step": 150
32
+ },
33
+ {
34
+ "epoch": 0.06666666666666667,
35
+ "grad_norm": 631.6712646484375,
36
+ "learning_rate": 0.000399708326752494,
37
+ "loss": 1384.57703125,
38
+ "step": 200
39
+ },
40
+ {
41
+ "epoch": 0.08333333333333333,
42
+ "grad_norm": 805.9771728515625,
43
+ "learning_rate": 0.0003988102673898103,
44
+ "loss": 1370.675,
45
+ "step": 250
46
+ },
47
+ {
48
+ "epoch": 0.08333333333333333,
49
+ "eval_accuracy": 0.18672140762463343,
50
+ "eval_loss": 45.42939758300781,
51
+ "eval_runtime": 25.5972,
52
+ "eval_samples_per_second": 4.883,
53
+ "eval_steps_per_second": 0.156,
54
+ "step": 250
55
+ },
56
+ {
57
+ "epoch": 0.1,
58
+ "grad_norm": 653.5927124023438,
59
+ "learning_rate": 0.00039730842777928766,
60
+ "loss": 1365.14890625,
61
+ "step": 300
62
+ },
63
+ {
64
+ "epoch": 0.11666666666666667,
65
+ "grad_norm": 995.5695190429688,
66
+ "learning_rate": 0.0003952073689584629,
67
+ "loss": 1359.084375,
68
+ "step": 350
69
+ },
70
+ {
71
+ "epoch": 0.13333333333333333,
72
+ "grad_norm": 825.2788696289062,
73
+ "learning_rate": 0.00039251347177392016,
74
+ "loss": 1355.68390625,
75
+ "step": 400
76
+ },
77
+ {
78
+ "epoch": 0.15,
79
+ "grad_norm": 615.8878784179688,
80
+ "learning_rate": 0.00038923491750286954,
81
+ "loss": 1353.1375,
82
+ "step": 450
83
+ },
84
+ {
85
+ "epoch": 0.16666666666666666,
86
+ "grad_norm": 895.9261474609375,
87
+ "learning_rate": 0.00038538166300687717,
88
+ "loss": 1349.4459375,
89
+ "step": 500
90
+ },
91
+ {
92
+ "epoch": 0.16666666666666666,
93
+ "eval_accuracy": 0.19390224828934507,
94
+ "eval_loss": 44.819725036621094,
95
+ "eval_runtime": 7.3949,
96
+ "eval_samples_per_second": 16.904,
97
+ "eval_steps_per_second": 0.541,
98
+ "step": 500
99
+ }
100
+ ],
101
+ "logging_steps": 50,
102
+ "max_steps": 3000,
103
+ "num_input_tokens_seen": 0,
104
+ "num_train_epochs": 9223372036854775807,
105
+ "save_steps": 500,
106
+ "stateful_callbacks": {
107
+ "TrainerControl": {
108
+ "args": {
109
+ "should_epoch_stop": false,
110
+ "should_evaluate": false,
111
+ "should_log": false,
112
+ "should_save": true,
113
+ "should_training_stop": false
114
+ },
115
+ "attributes": {}
116
+ }
117
+ },
118
+ "total_flos": 0.0,
119
+ "train_batch_size": 4,
120
+ "trial_name": null,
121
+ "trial_params": null
122
+ }
last-checkpoint/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:cdf1bf7ff316b803830588f07fd42e9d70dbb919d664b54789baf8adc99f17cf
3
+ size 5265