CodeIsAbstract commited on
Commit
c1dbe28
·
verified ·
1 Parent(s): a0ad4b3

Training in progress, step 20, checkpoint

Browse files
last-checkpoint/config.json ADDED
@@ -0,0 +1,33 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "architectures": [
3
+ "HybridFourierLM"
4
+ ],
5
+ "bos_token_id": 0,
6
+ "dropout": 0.05,
7
+ "dtype": "float32",
8
+ "eos_token_id": 0,
9
+ "latent_dim": 768,
10
+ "layer_types": [
11
+ "linear",
12
+ "linear",
13
+ "linear",
14
+ "softmax",
15
+ "linear",
16
+ "linear",
17
+ "linear",
18
+ "softmax",
19
+ "linear",
20
+ "linear",
21
+ "linear",
22
+ "softmax"
23
+ ],
24
+ "model_type": "hybrid_fourier_lm",
25
+ "num_layers": 12,
26
+ "num_modes": 64,
27
+ "pad_token_id": 1,
28
+ "tie_word_embeddings": true,
29
+ "time_scale": 128.0,
30
+ "transformers_version": "5.13.0",
31
+ "use_cache": false,
32
+ "vocab_size": 50277
33
+ }
last-checkpoint/model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:109a1ff562e60b4a3659de3f31e49413def647fff60dd801ec9ed38b39374255
3
+ size 579824888
last-checkpoint/optimizer.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:b8cdb8b10ce287047c72839d766120849b2d9aacf931ff44a1b8510116bdc7e6
3
+ size 1159794763
last-checkpoint/rng_state.pth ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:f0a81ce99965f5650998c3fdf0932f734f4f0b6b9b4f4bc0d128c8b5f65e9ca4
3
+ size 14645
last-checkpoint/scheduler.pt ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a7ad50ba85de2f08235ac51873a9e6e650ceaf5d93913d1c853bc575c8052ec1
3
+ size 1465
last-checkpoint/tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
last-checkpoint/tokenizer_config.json ADDED
@@ -0,0 +1,14 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "add_prefix_space": false,
3
+ "backend": "tokenizers",
4
+ "bos_token": "<|endoftext|>",
5
+ "eos_token": "<|endoftext|>",
6
+ "errors": "replace",
7
+ "is_local": false,
8
+ "local_files_only": false,
9
+ "model_max_length": 1000000000,
10
+ "pad_token": "<|padding|>",
11
+ "tokenizer_class": "GPTNeoXTokenizer",
12
+ "trim_offsets": true,
13
+ "unk_token": "<|endoftext|>"
14
+ }
last-checkpoint/trainer_state.json ADDED
@@ -0,0 +1,52 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "best_global_step": null,
3
+ "best_metric": null,
4
+ "best_model_checkpoint": null,
5
+ "epoch": 0.0004,
6
+ "eval_steps": 10,
7
+ "global_step": 20,
8
+ "is_hyper_param_search": false,
9
+ "is_local_process_zero": true,
10
+ "is_world_process_zero": true,
11
+ "log_history": [
12
+ {
13
+ "epoch": 0.0002,
14
+ "eval_accuracy": 0.043638611041376174,
15
+ "eval_loss": 9.762117385864258,
16
+ "eval_runtime": 6.9234,
17
+ "eval_samples_per_second": 280.355,
18
+ "eval_steps_per_second": 17.621,
19
+ "step": 10
20
+ },
21
+ {
22
+ "epoch": 0.0004,
23
+ "eval_accuracy": 0.06910715419957232,
24
+ "eval_loss": 8.825312614440918,
25
+ "eval_runtime": 6.4973,
26
+ "eval_samples_per_second": 298.738,
27
+ "eval_steps_per_second": 18.777,
28
+ "step": 20
29
+ }
30
+ ],
31
+ "logging_steps": 100,
32
+ "max_steps": 50000,
33
+ "num_input_tokens_seen": 0,
34
+ "num_train_epochs": 9223372036854775807,
35
+ "save_steps": 20,
36
+ "stateful_callbacks": {
37
+ "TrainerControl": {
38
+ "args": {
39
+ "should_epoch_stop": false,
40
+ "should_evaluate": false,
41
+ "should_log": false,
42
+ "should_save": true,
43
+ "should_training_stop": false
44
+ },
45
+ "attributes": {}
46
+ }
47
+ },
48
+ "total_flos": 783992173363200.0,
49
+ "train_batch_size": 120,
50
+ "trial_name": null,
51
+ "trial_params": null
52
+ }
last-checkpoint/training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:873e1a4204ced4196d959a031f55bc9264b3965e59e7590a3c6d11d44d86d0dc
3
+ size 5265