Commit ·
1610cce
1
Parent(s): e95b402
Training in progress, epoch 4
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- model.safetensors +1 -1
- run-5/checkpoint-272/config.json +87 -0
- run-5/checkpoint-272/merges.txt +0 -0
- run-5/checkpoint-272/model.safetensors +3 -0
- run-5/checkpoint-272/optimizer.pt +3 -0
- run-5/checkpoint-272/rng_state.pth +3 -0
- run-5/checkpoint-272/scheduler.pt +3 -0
- run-5/checkpoint-272/special_tokens_map.json +15 -0
- run-5/checkpoint-272/tokenizer.json +0 -0
- run-5/checkpoint-272/tokenizer_config.json +58 -0
- run-5/checkpoint-272/trainer_state.json +42 -0
- run-5/checkpoint-272/training_args.bin +3 -0
- run-5/checkpoint-272/vocab.json +0 -0
- run-6/checkpoint-136/config.json +87 -0
- run-6/checkpoint-136/merges.txt +0 -0
- run-6/checkpoint-136/model.safetensors +3 -0
- run-6/checkpoint-136/optimizer.pt +3 -0
- run-6/checkpoint-136/rng_state.pth +3 -0
- run-6/checkpoint-136/scheduler.pt +3 -0
- run-6/checkpoint-136/special_tokens_map.json +15 -0
- run-6/checkpoint-136/tokenizer.json +0 -0
- run-6/checkpoint-136/tokenizer_config.json +58 -0
- run-6/checkpoint-136/trainer_state.json +42 -0
- run-6/checkpoint-136/training_args.bin +3 -0
- run-6/checkpoint-136/vocab.json +0 -0
- run-6/checkpoint-204/config.json +87 -0
- run-6/checkpoint-204/merges.txt +0 -0
- run-6/checkpoint-204/model.safetensors +3 -0
- run-6/checkpoint-204/optimizer.pt +3 -0
- run-6/checkpoint-204/rng_state.pth +3 -0
- run-6/checkpoint-204/scheduler.pt +3 -0
- run-6/checkpoint-204/special_tokens_map.json +15 -0
- run-6/checkpoint-204/tokenizer.json +0 -0
- run-6/checkpoint-204/tokenizer_config.json +58 -0
- run-6/checkpoint-204/trainer_state.json +51 -0
- run-6/checkpoint-204/training_args.bin +3 -0
- run-6/checkpoint-204/vocab.json +0 -0
- run-6/checkpoint-272/model.safetensors +1 -1
- run-6/checkpoint-272/optimizer.pt +1 -1
- run-6/checkpoint-272/rng_state.pth +1 -1
- run-6/checkpoint-272/scheduler.pt +1 -1
- run-6/checkpoint-272/tokenizer.json +2 -2
- run-6/checkpoint-272/tokenizer_config.json +1 -1
- run-6/checkpoint-272/trainer_state.json +39 -12
- run-6/checkpoint-272/training_args.bin +1 -1
- run-6/checkpoint-68/config.json +87 -0
- run-6/checkpoint-68/merges.txt +0 -0
- run-6/checkpoint-68/model.safetensors +3 -0
- run-6/checkpoint-68/optimizer.pt +3 -0
- run-6/checkpoint-68/rng_state.pth +3 -0
model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 498692800
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:24df4b297f9fa796d4993f1c3a655ab4472259e1580b2510586de3527689e7da
|
| 3 |
size 498692800
|
run-5/checkpoint-272/config.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "roberta-base",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"RobertaForSequenceClassification"
|
| 5 |
+
],
|
| 6 |
+
"attention_probs_dropout_prob": 0.1,
|
| 7 |
+
"bos_token_id": 0,
|
| 8 |
+
"classifier_dropout": null,
|
| 9 |
+
"eos_token_id": 2,
|
| 10 |
+
"hidden_act": "gelu",
|
| 11 |
+
"hidden_dropout_prob": 0.1,
|
| 12 |
+
"hidden_size": 768,
|
| 13 |
+
"id2label": {
|
| 14 |
+
"0": "LABEL_0",
|
| 15 |
+
"1": "LABEL_1",
|
| 16 |
+
"2": "LABEL_2",
|
| 17 |
+
"3": "LABEL_3",
|
| 18 |
+
"4": "LABEL_4",
|
| 19 |
+
"5": "LABEL_5",
|
| 20 |
+
"6": "LABEL_6",
|
| 21 |
+
"7": "LABEL_7",
|
| 22 |
+
"8": "LABEL_8",
|
| 23 |
+
"9": "LABEL_9",
|
| 24 |
+
"10": "LABEL_10",
|
| 25 |
+
"11": "LABEL_11",
|
| 26 |
+
"12": "LABEL_12",
|
| 27 |
+
"13": "LABEL_13",
|
| 28 |
+
"14": "LABEL_14",
|
| 29 |
+
"15": "LABEL_15",
|
| 30 |
+
"16": "LABEL_16",
|
| 31 |
+
"17": "LABEL_17",
|
| 32 |
+
"18": "LABEL_18",
|
| 33 |
+
"19": "LABEL_19",
|
| 34 |
+
"20": "LABEL_20",
|
| 35 |
+
"21": "LABEL_21",
|
| 36 |
+
"22": "LABEL_22",
|
| 37 |
+
"23": "LABEL_23",
|
| 38 |
+
"24": "LABEL_24",
|
| 39 |
+
"25": "LABEL_25",
|
| 40 |
+
"26": "LABEL_26",
|
| 41 |
+
"27": "LABEL_27"
|
| 42 |
+
},
|
| 43 |
+
"initializer_range": 0.02,
|
| 44 |
+
"intermediate_size": 3072,
|
| 45 |
+
"label2id": {
|
| 46 |
+
"LABEL_0": 0,
|
| 47 |
+
"LABEL_1": 1,
|
| 48 |
+
"LABEL_10": 10,
|
| 49 |
+
"LABEL_11": 11,
|
| 50 |
+
"LABEL_12": 12,
|
| 51 |
+
"LABEL_13": 13,
|
| 52 |
+
"LABEL_14": 14,
|
| 53 |
+
"LABEL_15": 15,
|
| 54 |
+
"LABEL_16": 16,
|
| 55 |
+
"LABEL_17": 17,
|
| 56 |
+
"LABEL_18": 18,
|
| 57 |
+
"LABEL_19": 19,
|
| 58 |
+
"LABEL_2": 2,
|
| 59 |
+
"LABEL_20": 20,
|
| 60 |
+
"LABEL_21": 21,
|
| 61 |
+
"LABEL_22": 22,
|
| 62 |
+
"LABEL_23": 23,
|
| 63 |
+
"LABEL_24": 24,
|
| 64 |
+
"LABEL_25": 25,
|
| 65 |
+
"LABEL_26": 26,
|
| 66 |
+
"LABEL_27": 27,
|
| 67 |
+
"LABEL_3": 3,
|
| 68 |
+
"LABEL_4": 4,
|
| 69 |
+
"LABEL_5": 5,
|
| 70 |
+
"LABEL_6": 6,
|
| 71 |
+
"LABEL_7": 7,
|
| 72 |
+
"LABEL_8": 8,
|
| 73 |
+
"LABEL_9": 9
|
| 74 |
+
},
|
| 75 |
+
"layer_norm_eps": 1e-05,
|
| 76 |
+
"max_position_embeddings": 514,
|
| 77 |
+
"model_type": "roberta",
|
| 78 |
+
"num_attention_heads": 12,
|
| 79 |
+
"num_hidden_layers": 12,
|
| 80 |
+
"pad_token_id": 1,
|
| 81 |
+
"position_embedding_type": "absolute",
|
| 82 |
+
"torch_dtype": "float32",
|
| 83 |
+
"transformers_version": "4.35.2",
|
| 84 |
+
"type_vocab_size": 1,
|
| 85 |
+
"use_cache": true,
|
| 86 |
+
"vocab_size": 50265
|
| 87 |
+
}
|
run-5/checkpoint-272/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-5/checkpoint-272/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c0a33f8374f0b009669599ddd495aba375d673c547ceda3e1c14837ab4afa758
|
| 3 |
+
size 498692800
|
run-5/checkpoint-272/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:33ff2d2bfcfa54e192cf372585ea91c76ebb608da530976894325e4a16d9b212
|
| 3 |
+
size 997505402
|
run-5/checkpoint-272/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:823b0a05c87e760485d9d141a1d80d22ad6f3a0c9d14baa34c611af6732e5309
|
| 3 |
+
size 14244
|
run-5/checkpoint-272/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c950336249eeab44c29e3dd0cf06efdbc4e8502eb7c2d477aade0a9d5f13c345
|
| 3 |
+
size 1064
|
run-5/checkpoint-272/special_tokens_map.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": "<s>",
|
| 3 |
+
"cls_token": "<s>",
|
| 4 |
+
"eos_token": "</s>",
|
| 5 |
+
"mask_token": {
|
| 6 |
+
"content": "<mask>",
|
| 7 |
+
"lstrip": true,
|
| 8 |
+
"normalized": false,
|
| 9 |
+
"rstrip": false,
|
| 10 |
+
"single_word": false
|
| 11 |
+
},
|
| 12 |
+
"pad_token": "<pad>",
|
| 13 |
+
"sep_token": "</s>",
|
| 14 |
+
"unk_token": "<unk>"
|
| 15 |
+
}
|
run-5/checkpoint-272/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-5/checkpoint-272/tokenizer_config.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"0": {
|
| 5 |
+
"content": "<s>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": true,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"1": {
|
| 13 |
+
"content": "<pad>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": true,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"2": {
|
| 21 |
+
"content": "</s>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": true,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
},
|
| 28 |
+
"3": {
|
| 29 |
+
"content": "<unk>",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": true,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": true
|
| 35 |
+
},
|
| 36 |
+
"50264": {
|
| 37 |
+
"content": "<mask>",
|
| 38 |
+
"lstrip": true,
|
| 39 |
+
"normalized": false,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": true
|
| 43 |
+
}
|
| 44 |
+
},
|
| 45 |
+
"bos_token": "<s>",
|
| 46 |
+
"clean_up_tokenization_spaces": true,
|
| 47 |
+
"cls_token": "<s>",
|
| 48 |
+
"do_lower_case": false,
|
| 49 |
+
"eos_token": "</s>",
|
| 50 |
+
"errors": "replace",
|
| 51 |
+
"mask_token": "<mask>",
|
| 52 |
+
"model_max_length": 128,
|
| 53 |
+
"pad_token": "<pad>",
|
| 54 |
+
"sep_token": "</s>",
|
| 55 |
+
"tokenizer_class": "RobertaTokenizer",
|
| 56 |
+
"trim_offsets": true,
|
| 57 |
+
"unk_token": "<unk>"
|
| 58 |
+
}
|
run-5/checkpoint-272/trainer_state.json
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_metric": 0.0,
|
| 3 |
+
"best_model_checkpoint": "roberta-base-finetuned/run-5/checkpoint-136",
|
| 4 |
+
"epoch": 2.0,
|
| 5 |
+
"eval_steps": 500,
|
| 6 |
+
"global_step": 272,
|
| 7 |
+
"is_hyper_param_search": true,
|
| 8 |
+
"is_local_process_zero": true,
|
| 9 |
+
"is_world_process_zero": true,
|
| 10 |
+
"log_history": [
|
| 11 |
+
{
|
| 12 |
+
"epoch": 1.0,
|
| 13 |
+
"eval_f1": 0.0,
|
| 14 |
+
"eval_loss": 0.1628250777721405,
|
| 15 |
+
"eval_runtime": 1.768,
|
| 16 |
+
"eval_samples_per_second": 153.277,
|
| 17 |
+
"eval_steps_per_second": 1.131,
|
| 18 |
+
"step": 136
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"epoch": 2.0,
|
| 22 |
+
"eval_f1": 0.0,
|
| 23 |
+
"eval_loss": 0.15479044616222382,
|
| 24 |
+
"eval_runtime": 1.7694,
|
| 25 |
+
"eval_samples_per_second": 153.156,
|
| 26 |
+
"eval_steps_per_second": 1.13,
|
| 27 |
+
"step": 272
|
| 28 |
+
}
|
| 29 |
+
],
|
| 30 |
+
"logging_steps": 500,
|
| 31 |
+
"max_steps": 272,
|
| 32 |
+
"num_train_epochs": 2,
|
| 33 |
+
"save_steps": 500,
|
| 34 |
+
"total_flos": 0,
|
| 35 |
+
"trial_name": null,
|
| 36 |
+
"trial_params": {
|
| 37 |
+
"learning_rate": 2.4061789637885192e-05,
|
| 38 |
+
"num_train_epochs": 2,
|
| 39 |
+
"per_device_train_batch_size": 16,
|
| 40 |
+
"seed": 5
|
| 41 |
+
}
|
| 42 |
+
}
|
run-5/checkpoint-272/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b5e561920b0ab584c535b322f98226660fd2c79e91313f6c5e56d4ca00280fc3
|
| 3 |
+
size 4600
|
run-5/checkpoint-272/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-136/config.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "roberta-base",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"RobertaForSequenceClassification"
|
| 5 |
+
],
|
| 6 |
+
"attention_probs_dropout_prob": 0.1,
|
| 7 |
+
"bos_token_id": 0,
|
| 8 |
+
"classifier_dropout": null,
|
| 9 |
+
"eos_token_id": 2,
|
| 10 |
+
"hidden_act": "gelu",
|
| 11 |
+
"hidden_dropout_prob": 0.1,
|
| 12 |
+
"hidden_size": 768,
|
| 13 |
+
"id2label": {
|
| 14 |
+
"0": "LABEL_0",
|
| 15 |
+
"1": "LABEL_1",
|
| 16 |
+
"2": "LABEL_2",
|
| 17 |
+
"3": "LABEL_3",
|
| 18 |
+
"4": "LABEL_4",
|
| 19 |
+
"5": "LABEL_5",
|
| 20 |
+
"6": "LABEL_6",
|
| 21 |
+
"7": "LABEL_7",
|
| 22 |
+
"8": "LABEL_8",
|
| 23 |
+
"9": "LABEL_9",
|
| 24 |
+
"10": "LABEL_10",
|
| 25 |
+
"11": "LABEL_11",
|
| 26 |
+
"12": "LABEL_12",
|
| 27 |
+
"13": "LABEL_13",
|
| 28 |
+
"14": "LABEL_14",
|
| 29 |
+
"15": "LABEL_15",
|
| 30 |
+
"16": "LABEL_16",
|
| 31 |
+
"17": "LABEL_17",
|
| 32 |
+
"18": "LABEL_18",
|
| 33 |
+
"19": "LABEL_19",
|
| 34 |
+
"20": "LABEL_20",
|
| 35 |
+
"21": "LABEL_21",
|
| 36 |
+
"22": "LABEL_22",
|
| 37 |
+
"23": "LABEL_23",
|
| 38 |
+
"24": "LABEL_24",
|
| 39 |
+
"25": "LABEL_25",
|
| 40 |
+
"26": "LABEL_26",
|
| 41 |
+
"27": "LABEL_27"
|
| 42 |
+
},
|
| 43 |
+
"initializer_range": 0.02,
|
| 44 |
+
"intermediate_size": 3072,
|
| 45 |
+
"label2id": {
|
| 46 |
+
"LABEL_0": 0,
|
| 47 |
+
"LABEL_1": 1,
|
| 48 |
+
"LABEL_10": 10,
|
| 49 |
+
"LABEL_11": 11,
|
| 50 |
+
"LABEL_12": 12,
|
| 51 |
+
"LABEL_13": 13,
|
| 52 |
+
"LABEL_14": 14,
|
| 53 |
+
"LABEL_15": 15,
|
| 54 |
+
"LABEL_16": 16,
|
| 55 |
+
"LABEL_17": 17,
|
| 56 |
+
"LABEL_18": 18,
|
| 57 |
+
"LABEL_19": 19,
|
| 58 |
+
"LABEL_2": 2,
|
| 59 |
+
"LABEL_20": 20,
|
| 60 |
+
"LABEL_21": 21,
|
| 61 |
+
"LABEL_22": 22,
|
| 62 |
+
"LABEL_23": 23,
|
| 63 |
+
"LABEL_24": 24,
|
| 64 |
+
"LABEL_25": 25,
|
| 65 |
+
"LABEL_26": 26,
|
| 66 |
+
"LABEL_27": 27,
|
| 67 |
+
"LABEL_3": 3,
|
| 68 |
+
"LABEL_4": 4,
|
| 69 |
+
"LABEL_5": 5,
|
| 70 |
+
"LABEL_6": 6,
|
| 71 |
+
"LABEL_7": 7,
|
| 72 |
+
"LABEL_8": 8,
|
| 73 |
+
"LABEL_9": 9
|
| 74 |
+
},
|
| 75 |
+
"layer_norm_eps": 1e-05,
|
| 76 |
+
"max_position_embeddings": 514,
|
| 77 |
+
"model_type": "roberta",
|
| 78 |
+
"num_attention_heads": 12,
|
| 79 |
+
"num_hidden_layers": 12,
|
| 80 |
+
"pad_token_id": 1,
|
| 81 |
+
"position_embedding_type": "absolute",
|
| 82 |
+
"torch_dtype": "float32",
|
| 83 |
+
"transformers_version": "4.35.2",
|
| 84 |
+
"type_vocab_size": 1,
|
| 85 |
+
"use_cache": true,
|
| 86 |
+
"vocab_size": 50265
|
| 87 |
+
}
|
run-6/checkpoint-136/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-136/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:17d4f80f9793e83ad34de9d967d2991deab27b8e34c5def3d33e1594aaa84166
|
| 3 |
+
size 498692800
|
run-6/checkpoint-136/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f5824c696cdfd89fb76098526cfcac2176baa740714d67e3cff6ae2fc18ff372
|
| 3 |
+
size 997505402
|
run-6/checkpoint-136/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:48110f108e6a5c46c9507598e8bde0de764b16fafd8a07d876a0cf62355a332c
|
| 3 |
+
size 14308
|
run-6/checkpoint-136/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5c6a398aec080996e6a1b063886e28816554bddb4bb1071b4cf89cb31b63b27d
|
| 3 |
+
size 1064
|
run-6/checkpoint-136/special_tokens_map.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": "<s>",
|
| 3 |
+
"cls_token": "<s>",
|
| 4 |
+
"eos_token": "</s>",
|
| 5 |
+
"mask_token": {
|
| 6 |
+
"content": "<mask>",
|
| 7 |
+
"lstrip": true,
|
| 8 |
+
"normalized": false,
|
| 9 |
+
"rstrip": false,
|
| 10 |
+
"single_word": false
|
| 11 |
+
},
|
| 12 |
+
"pad_token": "<pad>",
|
| 13 |
+
"sep_token": "</s>",
|
| 14 |
+
"unk_token": "<unk>"
|
| 15 |
+
}
|
run-6/checkpoint-136/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-136/tokenizer_config.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"0": {
|
| 5 |
+
"content": "<s>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": true,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"1": {
|
| 13 |
+
"content": "<pad>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": true,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"2": {
|
| 21 |
+
"content": "</s>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": true,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
},
|
| 28 |
+
"3": {
|
| 29 |
+
"content": "<unk>",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": true,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": true
|
| 35 |
+
},
|
| 36 |
+
"50264": {
|
| 37 |
+
"content": "<mask>",
|
| 38 |
+
"lstrip": true,
|
| 39 |
+
"normalized": false,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": true
|
| 43 |
+
}
|
| 44 |
+
},
|
| 45 |
+
"bos_token": "<s>",
|
| 46 |
+
"clean_up_tokenization_spaces": true,
|
| 47 |
+
"cls_token": "<s>",
|
| 48 |
+
"do_lower_case": false,
|
| 49 |
+
"eos_token": "</s>",
|
| 50 |
+
"errors": "replace",
|
| 51 |
+
"mask_token": "<mask>",
|
| 52 |
+
"model_max_length": 128,
|
| 53 |
+
"pad_token": "<pad>",
|
| 54 |
+
"sep_token": "</s>",
|
| 55 |
+
"tokenizer_class": "RobertaTokenizer",
|
| 56 |
+
"trim_offsets": true,
|
| 57 |
+
"unk_token": "<unk>"
|
| 58 |
+
}
|
run-6/checkpoint-136/trainer_state.json
ADDED
|
@@ -0,0 +1,42 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_metric": 0.0,
|
| 3 |
+
"best_model_checkpoint": "roberta-base-finetuned/run-6/checkpoint-68",
|
| 4 |
+
"epoch": 2.0,
|
| 5 |
+
"eval_steps": 500,
|
| 6 |
+
"global_step": 136,
|
| 7 |
+
"is_hyper_param_search": true,
|
| 8 |
+
"is_local_process_zero": true,
|
| 9 |
+
"is_world_process_zero": true,
|
| 10 |
+
"log_history": [
|
| 11 |
+
{
|
| 12 |
+
"epoch": 1.0,
|
| 13 |
+
"eval_f1": 0.0,
|
| 14 |
+
"eval_loss": 0.27623069286346436,
|
| 15 |
+
"eval_runtime": 1.7887,
|
| 16 |
+
"eval_samples_per_second": 151.505,
|
| 17 |
+
"eval_steps_per_second": 1.118,
|
| 18 |
+
"step": 68
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"epoch": 2.0,
|
| 22 |
+
"eval_f1": 0.0,
|
| 23 |
+
"eval_loss": 0.21532878279685974,
|
| 24 |
+
"eval_runtime": 1.7558,
|
| 25 |
+
"eval_samples_per_second": 154.348,
|
| 26 |
+
"eval_steps_per_second": 1.139,
|
| 27 |
+
"step": 136
|
| 28 |
+
}
|
| 29 |
+
],
|
| 30 |
+
"logging_steps": 500,
|
| 31 |
+
"max_steps": 340,
|
| 32 |
+
"num_train_epochs": 5,
|
| 33 |
+
"save_steps": 500,
|
| 34 |
+
"total_flos": 0,
|
| 35 |
+
"trial_name": null,
|
| 36 |
+
"trial_params": {
|
| 37 |
+
"learning_rate": 9.656231101012835e-06,
|
| 38 |
+
"num_train_epochs": 5,
|
| 39 |
+
"per_device_train_batch_size": 32,
|
| 40 |
+
"seed": 13
|
| 41 |
+
}
|
| 42 |
+
}
|
run-6/checkpoint-136/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a125737cc9b04c4061972a35907d68d87488b05a430e52db5906b8a58739987f
|
| 3 |
+
size 4600
|
run-6/checkpoint-136/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-204/config.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "roberta-base",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"RobertaForSequenceClassification"
|
| 5 |
+
],
|
| 6 |
+
"attention_probs_dropout_prob": 0.1,
|
| 7 |
+
"bos_token_id": 0,
|
| 8 |
+
"classifier_dropout": null,
|
| 9 |
+
"eos_token_id": 2,
|
| 10 |
+
"hidden_act": "gelu",
|
| 11 |
+
"hidden_dropout_prob": 0.1,
|
| 12 |
+
"hidden_size": 768,
|
| 13 |
+
"id2label": {
|
| 14 |
+
"0": "LABEL_0",
|
| 15 |
+
"1": "LABEL_1",
|
| 16 |
+
"2": "LABEL_2",
|
| 17 |
+
"3": "LABEL_3",
|
| 18 |
+
"4": "LABEL_4",
|
| 19 |
+
"5": "LABEL_5",
|
| 20 |
+
"6": "LABEL_6",
|
| 21 |
+
"7": "LABEL_7",
|
| 22 |
+
"8": "LABEL_8",
|
| 23 |
+
"9": "LABEL_9",
|
| 24 |
+
"10": "LABEL_10",
|
| 25 |
+
"11": "LABEL_11",
|
| 26 |
+
"12": "LABEL_12",
|
| 27 |
+
"13": "LABEL_13",
|
| 28 |
+
"14": "LABEL_14",
|
| 29 |
+
"15": "LABEL_15",
|
| 30 |
+
"16": "LABEL_16",
|
| 31 |
+
"17": "LABEL_17",
|
| 32 |
+
"18": "LABEL_18",
|
| 33 |
+
"19": "LABEL_19",
|
| 34 |
+
"20": "LABEL_20",
|
| 35 |
+
"21": "LABEL_21",
|
| 36 |
+
"22": "LABEL_22",
|
| 37 |
+
"23": "LABEL_23",
|
| 38 |
+
"24": "LABEL_24",
|
| 39 |
+
"25": "LABEL_25",
|
| 40 |
+
"26": "LABEL_26",
|
| 41 |
+
"27": "LABEL_27"
|
| 42 |
+
},
|
| 43 |
+
"initializer_range": 0.02,
|
| 44 |
+
"intermediate_size": 3072,
|
| 45 |
+
"label2id": {
|
| 46 |
+
"LABEL_0": 0,
|
| 47 |
+
"LABEL_1": 1,
|
| 48 |
+
"LABEL_10": 10,
|
| 49 |
+
"LABEL_11": 11,
|
| 50 |
+
"LABEL_12": 12,
|
| 51 |
+
"LABEL_13": 13,
|
| 52 |
+
"LABEL_14": 14,
|
| 53 |
+
"LABEL_15": 15,
|
| 54 |
+
"LABEL_16": 16,
|
| 55 |
+
"LABEL_17": 17,
|
| 56 |
+
"LABEL_18": 18,
|
| 57 |
+
"LABEL_19": 19,
|
| 58 |
+
"LABEL_2": 2,
|
| 59 |
+
"LABEL_20": 20,
|
| 60 |
+
"LABEL_21": 21,
|
| 61 |
+
"LABEL_22": 22,
|
| 62 |
+
"LABEL_23": 23,
|
| 63 |
+
"LABEL_24": 24,
|
| 64 |
+
"LABEL_25": 25,
|
| 65 |
+
"LABEL_26": 26,
|
| 66 |
+
"LABEL_27": 27,
|
| 67 |
+
"LABEL_3": 3,
|
| 68 |
+
"LABEL_4": 4,
|
| 69 |
+
"LABEL_5": 5,
|
| 70 |
+
"LABEL_6": 6,
|
| 71 |
+
"LABEL_7": 7,
|
| 72 |
+
"LABEL_8": 8,
|
| 73 |
+
"LABEL_9": 9
|
| 74 |
+
},
|
| 75 |
+
"layer_norm_eps": 1e-05,
|
| 76 |
+
"max_position_embeddings": 514,
|
| 77 |
+
"model_type": "roberta",
|
| 78 |
+
"num_attention_heads": 12,
|
| 79 |
+
"num_hidden_layers": 12,
|
| 80 |
+
"pad_token_id": 1,
|
| 81 |
+
"position_embedding_type": "absolute",
|
| 82 |
+
"torch_dtype": "float32",
|
| 83 |
+
"transformers_version": "4.35.2",
|
| 84 |
+
"type_vocab_size": 1,
|
| 85 |
+
"use_cache": true,
|
| 86 |
+
"vocab_size": 50265
|
| 87 |
+
}
|
run-6/checkpoint-204/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-204/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:951368aedb90f2b811cb85753dcdd207637783c1f31aa00810f63fa3bccf9fd0
|
| 3 |
+
size 498692800
|
run-6/checkpoint-204/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dd76bf2a83f994330e0134ce397067df011e5be2eb2d9b9f79d380b47de4040d
|
| 3 |
+
size 997505402
|
run-6/checkpoint-204/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ebc14a5b892c07b34237a7c212063297e2616310add1966fef13fefce9e62437
|
| 3 |
+
size 14308
|
run-6/checkpoint-204/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:f1691de8e5d89384b5a01a9060f1d7c0c59dd0da9b4ea9811b0f1dbc7dc7e349
|
| 3 |
+
size 1064
|
run-6/checkpoint-204/special_tokens_map.json
ADDED
|
@@ -0,0 +1,15 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"bos_token": "<s>",
|
| 3 |
+
"cls_token": "<s>",
|
| 4 |
+
"eos_token": "</s>",
|
| 5 |
+
"mask_token": {
|
| 6 |
+
"content": "<mask>",
|
| 7 |
+
"lstrip": true,
|
| 8 |
+
"normalized": false,
|
| 9 |
+
"rstrip": false,
|
| 10 |
+
"single_word": false
|
| 11 |
+
},
|
| 12 |
+
"pad_token": "<pad>",
|
| 13 |
+
"sep_token": "</s>",
|
| 14 |
+
"unk_token": "<unk>"
|
| 15 |
+
}
|
run-6/checkpoint-204/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-204/tokenizer_config.json
ADDED
|
@@ -0,0 +1,58 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"add_prefix_space": false,
|
| 3 |
+
"added_tokens_decoder": {
|
| 4 |
+
"0": {
|
| 5 |
+
"content": "<s>",
|
| 6 |
+
"lstrip": false,
|
| 7 |
+
"normalized": true,
|
| 8 |
+
"rstrip": false,
|
| 9 |
+
"single_word": false,
|
| 10 |
+
"special": true
|
| 11 |
+
},
|
| 12 |
+
"1": {
|
| 13 |
+
"content": "<pad>",
|
| 14 |
+
"lstrip": false,
|
| 15 |
+
"normalized": true,
|
| 16 |
+
"rstrip": false,
|
| 17 |
+
"single_word": false,
|
| 18 |
+
"special": true
|
| 19 |
+
},
|
| 20 |
+
"2": {
|
| 21 |
+
"content": "</s>",
|
| 22 |
+
"lstrip": false,
|
| 23 |
+
"normalized": true,
|
| 24 |
+
"rstrip": false,
|
| 25 |
+
"single_word": false,
|
| 26 |
+
"special": true
|
| 27 |
+
},
|
| 28 |
+
"3": {
|
| 29 |
+
"content": "<unk>",
|
| 30 |
+
"lstrip": false,
|
| 31 |
+
"normalized": true,
|
| 32 |
+
"rstrip": false,
|
| 33 |
+
"single_word": false,
|
| 34 |
+
"special": true
|
| 35 |
+
},
|
| 36 |
+
"50264": {
|
| 37 |
+
"content": "<mask>",
|
| 38 |
+
"lstrip": true,
|
| 39 |
+
"normalized": false,
|
| 40 |
+
"rstrip": false,
|
| 41 |
+
"single_word": false,
|
| 42 |
+
"special": true
|
| 43 |
+
}
|
| 44 |
+
},
|
| 45 |
+
"bos_token": "<s>",
|
| 46 |
+
"clean_up_tokenization_spaces": true,
|
| 47 |
+
"cls_token": "<s>",
|
| 48 |
+
"do_lower_case": false,
|
| 49 |
+
"eos_token": "</s>",
|
| 50 |
+
"errors": "replace",
|
| 51 |
+
"mask_token": "<mask>",
|
| 52 |
+
"model_max_length": 128,
|
| 53 |
+
"pad_token": "<pad>",
|
| 54 |
+
"sep_token": "</s>",
|
| 55 |
+
"tokenizer_class": "RobertaTokenizer",
|
| 56 |
+
"trim_offsets": true,
|
| 57 |
+
"unk_token": "<unk>"
|
| 58 |
+
}
|
run-6/checkpoint-204/trainer_state.json
ADDED
|
@@ -0,0 +1,51 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_metric": 0.0,
|
| 3 |
+
"best_model_checkpoint": "roberta-base-finetuned/run-6/checkpoint-68",
|
| 4 |
+
"epoch": 3.0,
|
| 5 |
+
"eval_steps": 500,
|
| 6 |
+
"global_step": 204,
|
| 7 |
+
"is_hyper_param_search": true,
|
| 8 |
+
"is_local_process_zero": true,
|
| 9 |
+
"is_world_process_zero": true,
|
| 10 |
+
"log_history": [
|
| 11 |
+
{
|
| 12 |
+
"epoch": 1.0,
|
| 13 |
+
"eval_f1": 0.0,
|
| 14 |
+
"eval_loss": 0.27623069286346436,
|
| 15 |
+
"eval_runtime": 1.7887,
|
| 16 |
+
"eval_samples_per_second": 151.505,
|
| 17 |
+
"eval_steps_per_second": 1.118,
|
| 18 |
+
"step": 68
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"epoch": 2.0,
|
| 22 |
+
"eval_f1": 0.0,
|
| 23 |
+
"eval_loss": 0.21532878279685974,
|
| 24 |
+
"eval_runtime": 1.7558,
|
| 25 |
+
"eval_samples_per_second": 154.348,
|
| 26 |
+
"eval_steps_per_second": 1.139,
|
| 27 |
+
"step": 136
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"epoch": 3.0,
|
| 31 |
+
"eval_f1": 0.0,
|
| 32 |
+
"eval_loss": 0.19268403947353363,
|
| 33 |
+
"eval_runtime": 1.769,
|
| 34 |
+
"eval_samples_per_second": 153.196,
|
| 35 |
+
"eval_steps_per_second": 1.131,
|
| 36 |
+
"step": 204
|
| 37 |
+
}
|
| 38 |
+
],
|
| 39 |
+
"logging_steps": 500,
|
| 40 |
+
"max_steps": 340,
|
| 41 |
+
"num_train_epochs": 5,
|
| 42 |
+
"save_steps": 500,
|
| 43 |
+
"total_flos": 0,
|
| 44 |
+
"trial_name": null,
|
| 45 |
+
"trial_params": {
|
| 46 |
+
"learning_rate": 9.656231101012835e-06,
|
| 47 |
+
"num_train_epochs": 5,
|
| 48 |
+
"per_device_train_batch_size": 32,
|
| 49 |
+
"seed": 13
|
| 50 |
+
}
|
| 51 |
+
}
|
run-6/checkpoint-204/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a125737cc9b04c4061972a35907d68d87488b05a430e52db5906b8a58739987f
|
| 3 |
+
size 4600
|
run-6/checkpoint-204/vocab.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-272/model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 498692800
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:24df4b297f9fa796d4993f1c3a655ab4472259e1580b2510586de3527689e7da
|
| 3 |
size 498692800
|
run-6/checkpoint-272/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 997505402
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:363fbebfa284d8cc54fafd67b32758d29d648ac6ca9f5967ccfd1e82b8cab4b0
|
| 3 |
size 997505402
|
run-6/checkpoint-272/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14308
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c1395490277a30f2427a1fc55275bdbc591209f5a83c3f349e99a1f03b2975ce
|
| 3 |
size 14308
|
run-6/checkpoint-272/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1064
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:09b7d77263aa0a9bb24739a2c6e6452dfedbb4047ff897efcb64ba3ce4da6e2d
|
| 3 |
size 1064
|
run-6/checkpoint-272/tokenizer.json
CHANGED
|
@@ -2,13 +2,13 @@
|
|
| 2 |
"version": "1.0",
|
| 3 |
"truncation": {
|
| 4 |
"direction": "Right",
|
| 5 |
-
"max_length":
|
| 6 |
"strategy": "LongestFirst",
|
| 7 |
"stride": 0
|
| 8 |
},
|
| 9 |
"padding": {
|
| 10 |
"strategy": {
|
| 11 |
-
"Fixed":
|
| 12 |
},
|
| 13 |
"direction": "Right",
|
| 14 |
"pad_to_multiple_of": null,
|
|
|
|
| 2 |
"version": "1.0",
|
| 3 |
"truncation": {
|
| 4 |
"direction": "Right",
|
| 5 |
+
"max_length": 128,
|
| 6 |
"strategy": "LongestFirst",
|
| 7 |
"stride": 0
|
| 8 |
},
|
| 9 |
"padding": {
|
| 10 |
"strategy": {
|
| 11 |
+
"Fixed": 128
|
| 12 |
},
|
| 13 |
"direction": "Right",
|
| 14 |
"pad_to_multiple_of": null,
|
run-6/checkpoint-272/tokenizer_config.json
CHANGED
|
@@ -49,7 +49,7 @@
|
|
| 49 |
"eos_token": "</s>",
|
| 50 |
"errors": "replace",
|
| 51 |
"mask_token": "<mask>",
|
| 52 |
-
"model_max_length":
|
| 53 |
"pad_token": "<pad>",
|
| 54 |
"sep_token": "</s>",
|
| 55 |
"tokenizer_class": "RobertaTokenizer",
|
|
|
|
| 49 |
"eos_token": "</s>",
|
| 50 |
"errors": "replace",
|
| 51 |
"mask_token": "<mask>",
|
| 52 |
+
"model_max_length": 128,
|
| 53 |
"pad_token": "<pad>",
|
| 54 |
"sep_token": "</s>",
|
| 55 |
"tokenizer_class": "RobertaTokenizer",
|
run-6/checkpoint-272/trainer_state.json
CHANGED
|
@@ -1,7 +1,7 @@
|
|
| 1 |
{
|
| 2 |
"best_metric": 0.0,
|
| 3 |
-
"best_model_checkpoint": "roberta-base-finetuned/run-6/checkpoint-
|
| 4 |
-
"epoch":
|
| 5 |
"eval_steps": 500,
|
| 6 |
"global_step": 272,
|
| 7 |
"is_hyper_param_search": true,
|
|
@@ -11,23 +11,50 @@
|
|
| 11 |
{
|
| 12 |
"epoch": 1.0,
|
| 13 |
"eval_f1": 0.0,
|
| 14 |
-
"eval_loss": 0.
|
| 15 |
-
"eval_runtime":
|
| 16 |
-
"eval_samples_per_second":
|
| 17 |
-
"eval_steps_per_second":
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 18 |
"step": 272
|
| 19 |
}
|
| 20 |
],
|
| 21 |
"logging_steps": 500,
|
| 22 |
-
"max_steps":
|
| 23 |
-
"num_train_epochs":
|
| 24 |
"save_steps": 500,
|
| 25 |
"total_flos": 0,
|
| 26 |
"trial_name": null,
|
| 27 |
"trial_params": {
|
| 28 |
-
"learning_rate":
|
| 29 |
-
"num_train_epochs":
|
| 30 |
-
"per_device_train_batch_size":
|
| 31 |
-
"seed":
|
| 32 |
}
|
| 33 |
}
|
|
|
|
| 1 |
{
|
| 2 |
"best_metric": 0.0,
|
| 3 |
+
"best_model_checkpoint": "roberta-base-finetuned/run-6/checkpoint-68",
|
| 4 |
+
"epoch": 4.0,
|
| 5 |
"eval_steps": 500,
|
| 6 |
"global_step": 272,
|
| 7 |
"is_hyper_param_search": true,
|
|
|
|
| 11 |
{
|
| 12 |
"epoch": 1.0,
|
| 13 |
"eval_f1": 0.0,
|
| 14 |
+
"eval_loss": 0.27623069286346436,
|
| 15 |
+
"eval_runtime": 1.7887,
|
| 16 |
+
"eval_samples_per_second": 151.505,
|
| 17 |
+
"eval_steps_per_second": 1.118,
|
| 18 |
+
"step": 68
|
| 19 |
+
},
|
| 20 |
+
{
|
| 21 |
+
"epoch": 2.0,
|
| 22 |
+
"eval_f1": 0.0,
|
| 23 |
+
"eval_loss": 0.21532878279685974,
|
| 24 |
+
"eval_runtime": 1.7558,
|
| 25 |
+
"eval_samples_per_second": 154.348,
|
| 26 |
+
"eval_steps_per_second": 1.139,
|
| 27 |
+
"step": 136
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"epoch": 3.0,
|
| 31 |
+
"eval_f1": 0.0,
|
| 32 |
+
"eval_loss": 0.19268403947353363,
|
| 33 |
+
"eval_runtime": 1.769,
|
| 34 |
+
"eval_samples_per_second": 153.196,
|
| 35 |
+
"eval_steps_per_second": 1.131,
|
| 36 |
+
"step": 204
|
| 37 |
+
},
|
| 38 |
+
{
|
| 39 |
+
"epoch": 4.0,
|
| 40 |
+
"eval_f1": 0.0,
|
| 41 |
+
"eval_loss": 0.18309621512889862,
|
| 42 |
+
"eval_runtime": 1.7578,
|
| 43 |
+
"eval_samples_per_second": 154.173,
|
| 44 |
+
"eval_steps_per_second": 1.138,
|
| 45 |
"step": 272
|
| 46 |
}
|
| 47 |
],
|
| 48 |
"logging_steps": 500,
|
| 49 |
+
"max_steps": 340,
|
| 50 |
+
"num_train_epochs": 5,
|
| 51 |
"save_steps": 500,
|
| 52 |
"total_flos": 0,
|
| 53 |
"trial_name": null,
|
| 54 |
"trial_params": {
|
| 55 |
+
"learning_rate": 9.656231101012835e-06,
|
| 56 |
+
"num_train_epochs": 5,
|
| 57 |
+
"per_device_train_batch_size": 32,
|
| 58 |
+
"seed": 13
|
| 59 |
}
|
| 60 |
}
|
run-6/checkpoint-272/training_args.bin
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 4600
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:a125737cc9b04c4061972a35907d68d87488b05a430e52db5906b8a58739987f
|
| 3 |
size 4600
|
run-6/checkpoint-68/config.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"_name_or_path": "roberta-base",
|
| 3 |
+
"architectures": [
|
| 4 |
+
"RobertaForSequenceClassification"
|
| 5 |
+
],
|
| 6 |
+
"attention_probs_dropout_prob": 0.1,
|
| 7 |
+
"bos_token_id": 0,
|
| 8 |
+
"classifier_dropout": null,
|
| 9 |
+
"eos_token_id": 2,
|
| 10 |
+
"hidden_act": "gelu",
|
| 11 |
+
"hidden_dropout_prob": 0.1,
|
| 12 |
+
"hidden_size": 768,
|
| 13 |
+
"id2label": {
|
| 14 |
+
"0": "LABEL_0",
|
| 15 |
+
"1": "LABEL_1",
|
| 16 |
+
"2": "LABEL_2",
|
| 17 |
+
"3": "LABEL_3",
|
| 18 |
+
"4": "LABEL_4",
|
| 19 |
+
"5": "LABEL_5",
|
| 20 |
+
"6": "LABEL_6",
|
| 21 |
+
"7": "LABEL_7",
|
| 22 |
+
"8": "LABEL_8",
|
| 23 |
+
"9": "LABEL_9",
|
| 24 |
+
"10": "LABEL_10",
|
| 25 |
+
"11": "LABEL_11",
|
| 26 |
+
"12": "LABEL_12",
|
| 27 |
+
"13": "LABEL_13",
|
| 28 |
+
"14": "LABEL_14",
|
| 29 |
+
"15": "LABEL_15",
|
| 30 |
+
"16": "LABEL_16",
|
| 31 |
+
"17": "LABEL_17",
|
| 32 |
+
"18": "LABEL_18",
|
| 33 |
+
"19": "LABEL_19",
|
| 34 |
+
"20": "LABEL_20",
|
| 35 |
+
"21": "LABEL_21",
|
| 36 |
+
"22": "LABEL_22",
|
| 37 |
+
"23": "LABEL_23",
|
| 38 |
+
"24": "LABEL_24",
|
| 39 |
+
"25": "LABEL_25",
|
| 40 |
+
"26": "LABEL_26",
|
| 41 |
+
"27": "LABEL_27"
|
| 42 |
+
},
|
| 43 |
+
"initializer_range": 0.02,
|
| 44 |
+
"intermediate_size": 3072,
|
| 45 |
+
"label2id": {
|
| 46 |
+
"LABEL_0": 0,
|
| 47 |
+
"LABEL_1": 1,
|
| 48 |
+
"LABEL_10": 10,
|
| 49 |
+
"LABEL_11": 11,
|
| 50 |
+
"LABEL_12": 12,
|
| 51 |
+
"LABEL_13": 13,
|
| 52 |
+
"LABEL_14": 14,
|
| 53 |
+
"LABEL_15": 15,
|
| 54 |
+
"LABEL_16": 16,
|
| 55 |
+
"LABEL_17": 17,
|
| 56 |
+
"LABEL_18": 18,
|
| 57 |
+
"LABEL_19": 19,
|
| 58 |
+
"LABEL_2": 2,
|
| 59 |
+
"LABEL_20": 20,
|
| 60 |
+
"LABEL_21": 21,
|
| 61 |
+
"LABEL_22": 22,
|
| 62 |
+
"LABEL_23": 23,
|
| 63 |
+
"LABEL_24": 24,
|
| 64 |
+
"LABEL_25": 25,
|
| 65 |
+
"LABEL_26": 26,
|
| 66 |
+
"LABEL_27": 27,
|
| 67 |
+
"LABEL_3": 3,
|
| 68 |
+
"LABEL_4": 4,
|
| 69 |
+
"LABEL_5": 5,
|
| 70 |
+
"LABEL_6": 6,
|
| 71 |
+
"LABEL_7": 7,
|
| 72 |
+
"LABEL_8": 8,
|
| 73 |
+
"LABEL_9": 9
|
| 74 |
+
},
|
| 75 |
+
"layer_norm_eps": 1e-05,
|
| 76 |
+
"max_position_embeddings": 514,
|
| 77 |
+
"model_type": "roberta",
|
| 78 |
+
"num_attention_heads": 12,
|
| 79 |
+
"num_hidden_layers": 12,
|
| 80 |
+
"pad_token_id": 1,
|
| 81 |
+
"position_embedding_type": "absolute",
|
| 82 |
+
"torch_dtype": "float32",
|
| 83 |
+
"transformers_version": "4.35.2",
|
| 84 |
+
"type_vocab_size": 1,
|
| 85 |
+
"use_cache": true,
|
| 86 |
+
"vocab_size": 50265
|
| 87 |
+
}
|
run-6/checkpoint-68/merges.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
run-6/checkpoint-68/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:eb194a90f64cdadaf0548defc82f3f9d42bcb3e01e68c650558cc65aaafe26df
|
| 3 |
+
size 498692800
|
run-6/checkpoint-68/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:5de3f9a2fc39dc0aa541b3053ce17a42b08a2c097b88dabcf990b978add9b2e5
|
| 3 |
+
size 997505402
|
run-6/checkpoint-68/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:af50c8be9865745d31c3ed67983d1179439e59589353b5e68f077457eee7cccc
|
| 3 |
+
size 14308
|