Upload saved model files
Browse filesThis view is limited to 50 files because it contains too many changes. See raw diff
- checkpoint-141/config.json +33 -0
- checkpoint-141/model.safetensors +3 -0
- checkpoint-141/optimizer.pt +3 -0
- checkpoint-141/rng_state.pth +3 -0
- checkpoint-141/scaler.pt +3 -0
- checkpoint-141/scheduler.pt +3 -0
- checkpoint-141/special_tokens_map.json +37 -0
- checkpoint-141/tokenizer.json +0 -0
- checkpoint-141/tokenizer_config.json +56 -0
- checkpoint-141/trainer_state.json +87 -0
- checkpoint-141/training_args.bin +3 -0
- checkpoint-141/vocab.txt +0 -0
- checkpoint-170/config.json +33 -0
- checkpoint-170/model.safetensors +3 -0
- checkpoint-170/optimizer.pt +3 -0
- checkpoint-170/rng_state.pth +3 -0
- checkpoint-170/scaler.pt +3 -0
- checkpoint-170/scheduler.pt +3 -0
- checkpoint-170/special_tokens_map.json +37 -0
- checkpoint-170/tokenizer.json +0 -0
- checkpoint-170/tokenizer_config.json +56 -0
- checkpoint-170/trainer_state.json +84 -0
- checkpoint-170/training_args.bin +3 -0
- checkpoint-170/vocab.txt +0 -0
- checkpoint-188/config.json +33 -0
- checkpoint-188/model.safetensors +3 -0
- checkpoint-188/optimizer.pt +3 -0
- checkpoint-188/rng_state.pth +3 -0
- checkpoint-188/scaler.pt +3 -0
- checkpoint-188/scheduler.pt +3 -0
- checkpoint-188/special_tokens_map.json +37 -0
- checkpoint-188/tokenizer.json +0 -0
- checkpoint-188/tokenizer_config.json +56 -0
- checkpoint-188/trainer_state.json +104 -0
- checkpoint-188/training_args.bin +3 -0
- checkpoint-188/vocab.txt +0 -0
- checkpoint-235/config.json +33 -0
- checkpoint-235/model.safetensors +3 -0
- checkpoint-235/optimizer.pt +3 -0
- checkpoint-235/rng_state.pth +3 -0
- checkpoint-235/scaler.pt +3 -0
- checkpoint-235/scheduler.pt +3 -0
- checkpoint-235/special_tokens_map.json +37 -0
- checkpoint-235/tokenizer.json +0 -0
- checkpoint-235/tokenizer_config.json +56 -0
- checkpoint-235/trainer_state.json +121 -0
- checkpoint-235/training_args.bin +3 -0
- checkpoint-235/vocab.txt +0 -0
- checkpoint-255/config.json +33 -0
- checkpoint-255/model.safetensors +3 -0
checkpoint-141/config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"BertForSequenceClassification"
|
| 4 |
+
],
|
| 5 |
+
"attention_probs_dropout_prob": 0.1,
|
| 6 |
+
"classifier_dropout": null,
|
| 7 |
+
"gradient_checkpointing": false,
|
| 8 |
+
"hidden_act": "gelu",
|
| 9 |
+
"hidden_dropout_prob": 0.1,
|
| 10 |
+
"hidden_size": 768,
|
| 11 |
+
"id2label": {
|
| 12 |
+
"0": "TRAM",
|
| 13 |
+
"1": "ANNOCTR"
|
| 14 |
+
},
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 3072,
|
| 17 |
+
"label2id": {
|
| 18 |
+
"ANNOCTR": 1,
|
| 19 |
+
"TRAM": 0
|
| 20 |
+
},
|
| 21 |
+
"layer_norm_eps": 1e-12,
|
| 22 |
+
"max_position_embeddings": 512,
|
| 23 |
+
"model_type": "bert",
|
| 24 |
+
"num_attention_heads": 12,
|
| 25 |
+
"num_hidden_layers": 12,
|
| 26 |
+
"pad_token_id": 0,
|
| 27 |
+
"position_embedding_type": "absolute",
|
| 28 |
+
"torch_dtype": "float32",
|
| 29 |
+
"transformers_version": "4.55.2",
|
| 30 |
+
"type_vocab_size": 2,
|
| 31 |
+
"use_cache": true,
|
| 32 |
+
"vocab_size": 30522
|
| 33 |
+
}
|
checkpoint-141/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:db63af3e573bdbfe2715cf982cd9754a5353fbeedcd163802a1da4404cf99103
|
| 3 |
+
size 437958648
|
checkpoint-141/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b824a7da942e84fd7c8603c1a30ab34d59673b4547d77a264c78e068e86f6252
|
| 3 |
+
size 876038330
|
checkpoint-141/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:31d3c155bb3fb6994950ef7556da4ab88a9d77de972073b4673fa4ca32ceba95
|
| 3 |
+
size 14244
|
checkpoint-141/scaler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b30172cf14f5dbe00280d63e36224a9f28dc7a0e8b38a74ceb5eb284e84da363
|
| 3 |
+
size 988
|
checkpoint-141/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4ad952416cc9c4deaf0488ff0c747c00c0816e25082ec6ed973542521c8b2d69
|
| 3 |
+
size 1064
|
checkpoint-141/special_tokens_map.json
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cls_token": {
|
| 3 |
+
"content": "[CLS]",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"mask_token": {
|
| 10 |
+
"content": "[MASK]",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "[PAD]",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
},
|
| 23 |
+
"sep_token": {
|
| 24 |
+
"content": "[SEP]",
|
| 25 |
+
"lstrip": false,
|
| 26 |
+
"normalized": false,
|
| 27 |
+
"rstrip": false,
|
| 28 |
+
"single_word": false
|
| 29 |
+
},
|
| 30 |
+
"unk_token": {
|
| 31 |
+
"content": "[UNK]",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": false,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false
|
| 36 |
+
}
|
| 37 |
+
}
|
checkpoint-141/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-141/tokenizer_config.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"added_tokens_decoder": {
|
| 3 |
+
"0": {
|
| 4 |
+
"content": "[PAD]",
|
| 5 |
+
"lstrip": false,
|
| 6 |
+
"normalized": false,
|
| 7 |
+
"rstrip": false,
|
| 8 |
+
"single_word": false,
|
| 9 |
+
"special": true
|
| 10 |
+
},
|
| 11 |
+
"100": {
|
| 12 |
+
"content": "[UNK]",
|
| 13 |
+
"lstrip": false,
|
| 14 |
+
"normalized": false,
|
| 15 |
+
"rstrip": false,
|
| 16 |
+
"single_word": false,
|
| 17 |
+
"special": true
|
| 18 |
+
},
|
| 19 |
+
"101": {
|
| 20 |
+
"content": "[CLS]",
|
| 21 |
+
"lstrip": false,
|
| 22 |
+
"normalized": false,
|
| 23 |
+
"rstrip": false,
|
| 24 |
+
"single_word": false,
|
| 25 |
+
"special": true
|
| 26 |
+
},
|
| 27 |
+
"102": {
|
| 28 |
+
"content": "[SEP]",
|
| 29 |
+
"lstrip": false,
|
| 30 |
+
"normalized": false,
|
| 31 |
+
"rstrip": false,
|
| 32 |
+
"single_word": false,
|
| 33 |
+
"special": true
|
| 34 |
+
},
|
| 35 |
+
"103": {
|
| 36 |
+
"content": "[MASK]",
|
| 37 |
+
"lstrip": false,
|
| 38 |
+
"normalized": false,
|
| 39 |
+
"rstrip": false,
|
| 40 |
+
"single_word": false,
|
| 41 |
+
"special": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
"clean_up_tokenization_spaces": false,
|
| 45 |
+
"cls_token": "[CLS]",
|
| 46 |
+
"do_lower_case": true,
|
| 47 |
+
"extra_special_tokens": {},
|
| 48 |
+
"mask_token": "[MASK]",
|
| 49 |
+
"model_max_length": 512,
|
| 50 |
+
"pad_token": "[PAD]",
|
| 51 |
+
"sep_token": "[SEP]",
|
| 52 |
+
"strip_accents": null,
|
| 53 |
+
"tokenize_chinese_chars": true,
|
| 54 |
+
"tokenizer_class": "BertTokenizer",
|
| 55 |
+
"unk_token": "[UNK]"
|
| 56 |
+
}
|
checkpoint-141/trainer_state.json
ADDED
|
@@ -0,0 +1,87 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 141,
|
| 3 |
+
"best_metric": 0.8939393939393939,
|
| 4 |
+
"best_model_checkpoint": "./cysecbert-ttp-bert-router/checkpoint-141",
|
| 5 |
+
"epoch": 3.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 141,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 1.0,
|
| 14 |
+
"eval_accuracy": 0.8790322580645161,
|
| 15 |
+
"eval_f1": 0.88,
|
| 16 |
+
"eval_loss": 0.23156200349330902,
|
| 17 |
+
"eval_runtime": 0.5457,
|
| 18 |
+
"eval_samples_per_second": 454.439,
|
| 19 |
+
"eval_steps_per_second": 10.994,
|
| 20 |
+
"step": 47
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"epoch": 1.0638297872340425,
|
| 24 |
+
"grad_norm": 244324.0,
|
| 25 |
+
"learning_rate": 1.5829787234042555e-05,
|
| 26 |
+
"loss": 0.393,
|
| 27 |
+
"step": 50
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"epoch": 2.0,
|
| 31 |
+
"eval_accuracy": 0.8870967741935484,
|
| 32 |
+
"eval_f1": 0.889763779527559,
|
| 33 |
+
"eval_loss": 0.24126584827899933,
|
| 34 |
+
"eval_runtime": 0.5406,
|
| 35 |
+
"eval_samples_per_second": 458.709,
|
| 36 |
+
"eval_steps_per_second": 11.098,
|
| 37 |
+
"step": 94
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"epoch": 2.127659574468085,
|
| 41 |
+
"grad_norm": 360389.59375,
|
| 42 |
+
"learning_rate": 1.1574468085106382e-05,
|
| 43 |
+
"loss": 0.2003,
|
| 44 |
+
"step": 100
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"epoch": 3.0,
|
| 48 |
+
"eval_accuracy": 0.8870967741935484,
|
| 49 |
+
"eval_f1": 0.8939393939393939,
|
| 50 |
+
"eval_loss": 0.2832604646682739,
|
| 51 |
+
"eval_runtime": 0.5564,
|
| 52 |
+
"eval_samples_per_second": 445.721,
|
| 53 |
+
"eval_steps_per_second": 10.784,
|
| 54 |
+
"step": 141
|
| 55 |
+
}
|
| 56 |
+
],
|
| 57 |
+
"logging_steps": 50,
|
| 58 |
+
"max_steps": 235,
|
| 59 |
+
"num_input_tokens_seen": 0,
|
| 60 |
+
"num_train_epochs": 5,
|
| 61 |
+
"save_steps": 500,
|
| 62 |
+
"stateful_callbacks": {
|
| 63 |
+
"EarlyStoppingCallback": {
|
| 64 |
+
"args": {
|
| 65 |
+
"early_stopping_patience": 2,
|
| 66 |
+
"early_stopping_threshold": 0.0
|
| 67 |
+
},
|
| 68 |
+
"attributes": {
|
| 69 |
+
"early_stopping_patience_counter": 0
|
| 70 |
+
}
|
| 71 |
+
},
|
| 72 |
+
"TrainerControl": {
|
| 73 |
+
"args": {
|
| 74 |
+
"should_epoch_stop": false,
|
| 75 |
+
"should_evaluate": false,
|
| 76 |
+
"should_log": false,
|
| 77 |
+
"should_save": true,
|
| 78 |
+
"should_training_stop": false
|
| 79 |
+
},
|
| 80 |
+
"attributes": {}
|
| 81 |
+
}
|
| 82 |
+
},
|
| 83 |
+
"total_flos": 1752319628697600.0,
|
| 84 |
+
"train_batch_size": 48,
|
| 85 |
+
"trial_name": null,
|
| 86 |
+
"trial_params": null
|
| 87 |
+
}
|
checkpoint-141/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aa4f68038d34e18bca323801ebee2f977ae0e2c8a6e6a48bf186f9aaacac3db0
|
| 3 |
+
size 5368
|
checkpoint-141/vocab.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-170/config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"BertForSequenceClassification"
|
| 4 |
+
],
|
| 5 |
+
"attention_probs_dropout_prob": 0.1,
|
| 6 |
+
"classifier_dropout": null,
|
| 7 |
+
"gradient_checkpointing": false,
|
| 8 |
+
"hidden_act": "gelu",
|
| 9 |
+
"hidden_dropout_prob": 0.1,
|
| 10 |
+
"hidden_size": 768,
|
| 11 |
+
"id2label": {
|
| 12 |
+
"0": "TRAM",
|
| 13 |
+
"1": "ANNOCTR"
|
| 14 |
+
},
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 3072,
|
| 17 |
+
"label2id": {
|
| 18 |
+
"ANNOCTR": 1,
|
| 19 |
+
"TRAM": 0
|
| 20 |
+
},
|
| 21 |
+
"layer_norm_eps": 1e-12,
|
| 22 |
+
"max_position_embeddings": 512,
|
| 23 |
+
"model_type": "bert",
|
| 24 |
+
"num_attention_heads": 12,
|
| 25 |
+
"num_hidden_layers": 12,
|
| 26 |
+
"pad_token_id": 0,
|
| 27 |
+
"position_embedding_type": "absolute",
|
| 28 |
+
"torch_dtype": "float32",
|
| 29 |
+
"transformers_version": "4.55.2",
|
| 30 |
+
"type_vocab_size": 2,
|
| 31 |
+
"use_cache": true,
|
| 32 |
+
"vocab_size": 30522
|
| 33 |
+
}
|
checkpoint-170/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:39e180718fb1484ebd30e7d2d1562bbf3fea53e0edcc435064f32e1dd03e0204
|
| 3 |
+
size 437958648
|
checkpoint-170/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1eb0901e08072ef968b5eec4bcb9fd6532eb0a912af90912064a775d1f00cad8
|
| 3 |
+
size 876038330
|
checkpoint-170/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b4efe0f47241e6fc27a6d35b15954af11bc2804162f8a5fa82aaf9fe783c305f
|
| 3 |
+
size 14244
|
checkpoint-170/scaler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b30172cf14f5dbe00280d63e36224a9f28dc7a0e8b38a74ceb5eb284e84da363
|
| 3 |
+
size 988
|
checkpoint-170/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:c9ca867b5f07a5b0020195342eac2eb6841947051fea59ed1d774eee72ba43ae
|
| 3 |
+
size 1064
|
checkpoint-170/special_tokens_map.json
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cls_token": {
|
| 3 |
+
"content": "[CLS]",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"mask_token": {
|
| 10 |
+
"content": "[MASK]",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "[PAD]",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
},
|
| 23 |
+
"sep_token": {
|
| 24 |
+
"content": "[SEP]",
|
| 25 |
+
"lstrip": false,
|
| 26 |
+
"normalized": false,
|
| 27 |
+
"rstrip": false,
|
| 28 |
+
"single_word": false
|
| 29 |
+
},
|
| 30 |
+
"unk_token": {
|
| 31 |
+
"content": "[UNK]",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": false,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false
|
| 36 |
+
}
|
| 37 |
+
}
|
checkpoint-170/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-170/tokenizer_config.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"added_tokens_decoder": {
|
| 3 |
+
"0": {
|
| 4 |
+
"content": "[PAD]",
|
| 5 |
+
"lstrip": false,
|
| 6 |
+
"normalized": false,
|
| 7 |
+
"rstrip": false,
|
| 8 |
+
"single_word": false,
|
| 9 |
+
"special": true
|
| 10 |
+
},
|
| 11 |
+
"100": {
|
| 12 |
+
"content": "[UNK]",
|
| 13 |
+
"lstrip": false,
|
| 14 |
+
"normalized": false,
|
| 15 |
+
"rstrip": false,
|
| 16 |
+
"single_word": false,
|
| 17 |
+
"special": true
|
| 18 |
+
},
|
| 19 |
+
"101": {
|
| 20 |
+
"content": "[CLS]",
|
| 21 |
+
"lstrip": false,
|
| 22 |
+
"normalized": false,
|
| 23 |
+
"rstrip": false,
|
| 24 |
+
"single_word": false,
|
| 25 |
+
"special": true
|
| 26 |
+
},
|
| 27 |
+
"102": {
|
| 28 |
+
"content": "[SEP]",
|
| 29 |
+
"lstrip": false,
|
| 30 |
+
"normalized": false,
|
| 31 |
+
"rstrip": false,
|
| 32 |
+
"single_word": false,
|
| 33 |
+
"special": true
|
| 34 |
+
},
|
| 35 |
+
"103": {
|
| 36 |
+
"content": "[MASK]",
|
| 37 |
+
"lstrip": false,
|
| 38 |
+
"normalized": false,
|
| 39 |
+
"rstrip": false,
|
| 40 |
+
"single_word": false,
|
| 41 |
+
"special": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
"clean_up_tokenization_spaces": false,
|
| 45 |
+
"cls_token": "[CLS]",
|
| 46 |
+
"do_lower_case": true,
|
| 47 |
+
"extra_special_tokens": {},
|
| 48 |
+
"mask_token": "[MASK]",
|
| 49 |
+
"model_max_length": 512,
|
| 50 |
+
"pad_token": "[PAD]",
|
| 51 |
+
"sep_token": "[SEP]",
|
| 52 |
+
"strip_accents": null,
|
| 53 |
+
"tokenize_chinese_chars": true,
|
| 54 |
+
"tokenizer_class": "BertTokenizer",
|
| 55 |
+
"unk_token": "[UNK]"
|
| 56 |
+
}
|
checkpoint-170/trainer_state.json
ADDED
|
@@ -0,0 +1,84 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 170,
|
| 3 |
+
"best_metric": 0.8,
|
| 4 |
+
"best_model_checkpoint": "./cysecbert-ttp-bert-router/checkpoint-170",
|
| 5 |
+
"epoch": 2.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 170,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 0.5882352941176471,
|
| 14 |
+
"grad_norm": 195774.109375,
|
| 15 |
+
"learning_rate": 1.7694117647058826e-05,
|
| 16 |
+
"loss": 0.4698,
|
| 17 |
+
"step": 50
|
| 18 |
+
},
|
| 19 |
+
{
|
| 20 |
+
"epoch": 1.0,
|
| 21 |
+
"eval_accuracy": 0.8463251670378619,
|
| 22 |
+
"eval_f1": 0.7661016949152543,
|
| 23 |
+
"eval_loss": 0.29931211471557617,
|
| 24 |
+
"eval_runtime": 0.9434,
|
| 25 |
+
"eval_samples_per_second": 475.961,
|
| 26 |
+
"eval_steps_per_second": 10.6,
|
| 27 |
+
"step": 85
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"epoch": 1.1764705882352942,
|
| 31 |
+
"grad_norm": 215225.09375,
|
| 32 |
+
"learning_rate": 1.5341176470588238e-05,
|
| 33 |
+
"loss": 0.299,
|
| 34 |
+
"step": 100
|
| 35 |
+
},
|
| 36 |
+
{
|
| 37 |
+
"epoch": 1.7647058823529411,
|
| 38 |
+
"grad_norm": 529117.0625,
|
| 39 |
+
"learning_rate": 1.2988235294117649e-05,
|
| 40 |
+
"loss": 0.2068,
|
| 41 |
+
"step": 150
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"epoch": 2.0,
|
| 45 |
+
"eval_accuracy": 0.8775055679287305,
|
| 46 |
+
"eval_f1": 0.8,
|
| 47 |
+
"eval_loss": 0.24925704300403595,
|
| 48 |
+
"eval_runtime": 0.926,
|
| 49 |
+
"eval_samples_per_second": 484.876,
|
| 50 |
+
"eval_steps_per_second": 10.799,
|
| 51 |
+
"step": 170
|
| 52 |
+
}
|
| 53 |
+
],
|
| 54 |
+
"logging_steps": 50,
|
| 55 |
+
"max_steps": 425,
|
| 56 |
+
"num_input_tokens_seen": 0,
|
| 57 |
+
"num_train_epochs": 5,
|
| 58 |
+
"save_steps": 500,
|
| 59 |
+
"stateful_callbacks": {
|
| 60 |
+
"EarlyStoppingCallback": {
|
| 61 |
+
"args": {
|
| 62 |
+
"early_stopping_patience": 2,
|
| 63 |
+
"early_stopping_threshold": 0.0
|
| 64 |
+
},
|
| 65 |
+
"attributes": {
|
| 66 |
+
"early_stopping_patience_counter": 0
|
| 67 |
+
}
|
| 68 |
+
},
|
| 69 |
+
"TrainerControl": {
|
| 70 |
+
"args": {
|
| 71 |
+
"should_epoch_stop": false,
|
| 72 |
+
"should_evaluate": false,
|
| 73 |
+
"should_log": false,
|
| 74 |
+
"should_save": true,
|
| 75 |
+
"should_training_stop": false
|
| 76 |
+
},
|
| 77 |
+
"attributes": {}
|
| 78 |
+
}
|
| 79 |
+
},
|
| 80 |
+
"total_flos": 2124358660976640.0,
|
| 81 |
+
"train_batch_size": 48,
|
| 82 |
+
"trial_name": null,
|
| 83 |
+
"trial_params": null
|
| 84 |
+
}
|
checkpoint-170/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2eb49e4ddefef59ed0e951ac7dbde7059171d7f9b6de6437c5177c6ce38bcc3d
|
| 3 |
+
size 5368
|
checkpoint-170/vocab.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-188/config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"BertForSequenceClassification"
|
| 4 |
+
],
|
| 5 |
+
"attention_probs_dropout_prob": 0.1,
|
| 6 |
+
"classifier_dropout": null,
|
| 7 |
+
"gradient_checkpointing": false,
|
| 8 |
+
"hidden_act": "gelu",
|
| 9 |
+
"hidden_dropout_prob": 0.1,
|
| 10 |
+
"hidden_size": 768,
|
| 11 |
+
"id2label": {
|
| 12 |
+
"0": "TRAM",
|
| 13 |
+
"1": "ANNOCTR"
|
| 14 |
+
},
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 3072,
|
| 17 |
+
"label2id": {
|
| 18 |
+
"ANNOCTR": 1,
|
| 19 |
+
"TRAM": 0
|
| 20 |
+
},
|
| 21 |
+
"layer_norm_eps": 1e-12,
|
| 22 |
+
"max_position_embeddings": 512,
|
| 23 |
+
"model_type": "bert",
|
| 24 |
+
"num_attention_heads": 12,
|
| 25 |
+
"num_hidden_layers": 12,
|
| 26 |
+
"pad_token_id": 0,
|
| 27 |
+
"position_embedding_type": "absolute",
|
| 28 |
+
"torch_dtype": "float32",
|
| 29 |
+
"transformers_version": "4.55.2",
|
| 30 |
+
"type_vocab_size": 2,
|
| 31 |
+
"use_cache": true,
|
| 32 |
+
"vocab_size": 30522
|
| 33 |
+
}
|
checkpoint-188/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:52329fd83f15c93b643df934cee4acb8bf39b7474468dfd511e573ccfdab3b64
|
| 3 |
+
size 437958648
|
checkpoint-188/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aec8f1ee5a63f04b2ef14be5995968f39d58b110e97e578bb5505bbfff2c86ac
|
| 3 |
+
size 876038330
|
checkpoint-188/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:2364fa4dc0aceb9687fc6bcfc81d0cbcf0a4e9e920964e68971cafa1f7395554
|
| 3 |
+
size 14244
|
checkpoint-188/scaler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b30172cf14f5dbe00280d63e36224a9f28dc7a0e8b38a74ceb5eb284e84da363
|
| 3 |
+
size 988
|
checkpoint-188/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:24540af78f05e096d4d2f1a22a5efa5aa30308c905b0ca3d67079dece6935d90
|
| 3 |
+
size 1064
|
checkpoint-188/special_tokens_map.json
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cls_token": {
|
| 3 |
+
"content": "[CLS]",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"mask_token": {
|
| 10 |
+
"content": "[MASK]",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "[PAD]",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
},
|
| 23 |
+
"sep_token": {
|
| 24 |
+
"content": "[SEP]",
|
| 25 |
+
"lstrip": false,
|
| 26 |
+
"normalized": false,
|
| 27 |
+
"rstrip": false,
|
| 28 |
+
"single_word": false
|
| 29 |
+
},
|
| 30 |
+
"unk_token": {
|
| 31 |
+
"content": "[UNK]",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": false,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false
|
| 36 |
+
}
|
| 37 |
+
}
|
checkpoint-188/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-188/tokenizer_config.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"added_tokens_decoder": {
|
| 3 |
+
"0": {
|
| 4 |
+
"content": "[PAD]",
|
| 5 |
+
"lstrip": false,
|
| 6 |
+
"normalized": false,
|
| 7 |
+
"rstrip": false,
|
| 8 |
+
"single_word": false,
|
| 9 |
+
"special": true
|
| 10 |
+
},
|
| 11 |
+
"100": {
|
| 12 |
+
"content": "[UNK]",
|
| 13 |
+
"lstrip": false,
|
| 14 |
+
"normalized": false,
|
| 15 |
+
"rstrip": false,
|
| 16 |
+
"single_word": false,
|
| 17 |
+
"special": true
|
| 18 |
+
},
|
| 19 |
+
"101": {
|
| 20 |
+
"content": "[CLS]",
|
| 21 |
+
"lstrip": false,
|
| 22 |
+
"normalized": false,
|
| 23 |
+
"rstrip": false,
|
| 24 |
+
"single_word": false,
|
| 25 |
+
"special": true
|
| 26 |
+
},
|
| 27 |
+
"102": {
|
| 28 |
+
"content": "[SEP]",
|
| 29 |
+
"lstrip": false,
|
| 30 |
+
"normalized": false,
|
| 31 |
+
"rstrip": false,
|
| 32 |
+
"single_word": false,
|
| 33 |
+
"special": true
|
| 34 |
+
},
|
| 35 |
+
"103": {
|
| 36 |
+
"content": "[MASK]",
|
| 37 |
+
"lstrip": false,
|
| 38 |
+
"normalized": false,
|
| 39 |
+
"rstrip": false,
|
| 40 |
+
"single_word": false,
|
| 41 |
+
"special": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
"clean_up_tokenization_spaces": false,
|
| 45 |
+
"cls_token": "[CLS]",
|
| 46 |
+
"do_lower_case": true,
|
| 47 |
+
"extra_special_tokens": {},
|
| 48 |
+
"mask_token": "[MASK]",
|
| 49 |
+
"model_max_length": 512,
|
| 50 |
+
"pad_token": "[PAD]",
|
| 51 |
+
"sep_token": "[SEP]",
|
| 52 |
+
"strip_accents": null,
|
| 53 |
+
"tokenize_chinese_chars": true,
|
| 54 |
+
"tokenizer_class": "BertTokenizer",
|
| 55 |
+
"unk_token": "[UNK]"
|
| 56 |
+
}
|
checkpoint-188/trainer_state.json
ADDED
|
@@ -0,0 +1,104 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 141,
|
| 3 |
+
"best_metric": 0.8939393939393939,
|
| 4 |
+
"best_model_checkpoint": "./cysecbert-ttp-bert-router/checkpoint-141",
|
| 5 |
+
"epoch": 4.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 188,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 1.0,
|
| 14 |
+
"eval_accuracy": 0.8790322580645161,
|
| 15 |
+
"eval_f1": 0.88,
|
| 16 |
+
"eval_loss": 0.23156200349330902,
|
| 17 |
+
"eval_runtime": 0.5457,
|
| 18 |
+
"eval_samples_per_second": 454.439,
|
| 19 |
+
"eval_steps_per_second": 10.994,
|
| 20 |
+
"step": 47
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"epoch": 1.0638297872340425,
|
| 24 |
+
"grad_norm": 244324.0,
|
| 25 |
+
"learning_rate": 1.5829787234042555e-05,
|
| 26 |
+
"loss": 0.393,
|
| 27 |
+
"step": 50
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"epoch": 2.0,
|
| 31 |
+
"eval_accuracy": 0.8870967741935484,
|
| 32 |
+
"eval_f1": 0.889763779527559,
|
| 33 |
+
"eval_loss": 0.24126584827899933,
|
| 34 |
+
"eval_runtime": 0.5406,
|
| 35 |
+
"eval_samples_per_second": 458.709,
|
| 36 |
+
"eval_steps_per_second": 11.098,
|
| 37 |
+
"step": 94
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"epoch": 2.127659574468085,
|
| 41 |
+
"grad_norm": 360389.59375,
|
| 42 |
+
"learning_rate": 1.1574468085106382e-05,
|
| 43 |
+
"loss": 0.2003,
|
| 44 |
+
"step": 100
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"epoch": 3.0,
|
| 48 |
+
"eval_accuracy": 0.8870967741935484,
|
| 49 |
+
"eval_f1": 0.8939393939393939,
|
| 50 |
+
"eval_loss": 0.2832604646682739,
|
| 51 |
+
"eval_runtime": 0.5564,
|
| 52 |
+
"eval_samples_per_second": 445.721,
|
| 53 |
+
"eval_steps_per_second": 10.784,
|
| 54 |
+
"step": 141
|
| 55 |
+
},
|
| 56 |
+
{
|
| 57 |
+
"epoch": 3.1914893617021276,
|
| 58 |
+
"grad_norm": 309358.59375,
|
| 59 |
+
"learning_rate": 7.3191489361702125e-06,
|
| 60 |
+
"loss": 0.1351,
|
| 61 |
+
"step": 150
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"epoch": 4.0,
|
| 65 |
+
"eval_accuracy": 0.8951612903225806,
|
| 66 |
+
"eval_f1": 0.8916666666666667,
|
| 67 |
+
"eval_loss": 0.2920267879962921,
|
| 68 |
+
"eval_runtime": 0.5436,
|
| 69 |
+
"eval_samples_per_second": 456.231,
|
| 70 |
+
"eval_steps_per_second": 11.038,
|
| 71 |
+
"step": 188
|
| 72 |
+
}
|
| 73 |
+
],
|
| 74 |
+
"logging_steps": 50,
|
| 75 |
+
"max_steps": 235,
|
| 76 |
+
"num_input_tokens_seen": 0,
|
| 77 |
+
"num_train_epochs": 5,
|
| 78 |
+
"save_steps": 500,
|
| 79 |
+
"stateful_callbacks": {
|
| 80 |
+
"EarlyStoppingCallback": {
|
| 81 |
+
"args": {
|
| 82 |
+
"early_stopping_patience": 2,
|
| 83 |
+
"early_stopping_threshold": 0.0
|
| 84 |
+
},
|
| 85 |
+
"attributes": {
|
| 86 |
+
"early_stopping_patience_counter": 1
|
| 87 |
+
}
|
| 88 |
+
},
|
| 89 |
+
"TrainerControl": {
|
| 90 |
+
"args": {
|
| 91 |
+
"should_epoch_stop": false,
|
| 92 |
+
"should_evaluate": false,
|
| 93 |
+
"should_log": false,
|
| 94 |
+
"should_save": true,
|
| 95 |
+
"should_training_stop": false
|
| 96 |
+
},
|
| 97 |
+
"attributes": {}
|
| 98 |
+
}
|
| 99 |
+
},
|
| 100 |
+
"total_flos": 2336426171596800.0,
|
| 101 |
+
"train_batch_size": 48,
|
| 102 |
+
"trial_name": null,
|
| 103 |
+
"trial_params": null
|
| 104 |
+
}
|
checkpoint-188/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aa4f68038d34e18bca323801ebee2f977ae0e2c8a6e6a48bf186f9aaacac3db0
|
| 3 |
+
size 5368
|
checkpoint-188/vocab.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-235/config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"BertForSequenceClassification"
|
| 4 |
+
],
|
| 5 |
+
"attention_probs_dropout_prob": 0.1,
|
| 6 |
+
"classifier_dropout": null,
|
| 7 |
+
"gradient_checkpointing": false,
|
| 8 |
+
"hidden_act": "gelu",
|
| 9 |
+
"hidden_dropout_prob": 0.1,
|
| 10 |
+
"hidden_size": 768,
|
| 11 |
+
"id2label": {
|
| 12 |
+
"0": "TRAM",
|
| 13 |
+
"1": "ANNOCTR"
|
| 14 |
+
},
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 3072,
|
| 17 |
+
"label2id": {
|
| 18 |
+
"ANNOCTR": 1,
|
| 19 |
+
"TRAM": 0
|
| 20 |
+
},
|
| 21 |
+
"layer_norm_eps": 1e-12,
|
| 22 |
+
"max_position_embeddings": 512,
|
| 23 |
+
"model_type": "bert",
|
| 24 |
+
"num_attention_heads": 12,
|
| 25 |
+
"num_hidden_layers": 12,
|
| 26 |
+
"pad_token_id": 0,
|
| 27 |
+
"position_embedding_type": "absolute",
|
| 28 |
+
"torch_dtype": "float32",
|
| 29 |
+
"transformers_version": "4.55.2",
|
| 30 |
+
"type_vocab_size": 2,
|
| 31 |
+
"use_cache": true,
|
| 32 |
+
"vocab_size": 30522
|
| 33 |
+
}
|
checkpoint-235/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:44d8653d3afdb0baffe9cf4b45d2b823f40e7fde26c7e595e299c80967a1a2ba
|
| 3 |
+
size 437958648
|
checkpoint-235/optimizer.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:ad8757b126ab16096195c84c83037cce70a5e86f824cd0aebbac4474310b12e4
|
| 3 |
+
size 876038330
|
checkpoint-235/rng_state.pth
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:77f3152f72695410ca19230d791ca8e0b8bcaa727c6000020d57ae4a51439b97
|
| 3 |
+
size 14244
|
checkpoint-235/scaler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b30172cf14f5dbe00280d63e36224a9f28dc7a0e8b38a74ceb5eb284e84da363
|
| 3 |
+
size 988
|
checkpoint-235/scheduler.pt
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:83526710ecacebf5cbd106052ec21d8ce799f8bdf2ac8c540d73190260bb6224
|
| 3 |
+
size 1064
|
checkpoint-235/special_tokens_map.json
ADDED
|
@@ -0,0 +1,37 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"cls_token": {
|
| 3 |
+
"content": "[CLS]",
|
| 4 |
+
"lstrip": false,
|
| 5 |
+
"normalized": false,
|
| 6 |
+
"rstrip": false,
|
| 7 |
+
"single_word": false
|
| 8 |
+
},
|
| 9 |
+
"mask_token": {
|
| 10 |
+
"content": "[MASK]",
|
| 11 |
+
"lstrip": false,
|
| 12 |
+
"normalized": false,
|
| 13 |
+
"rstrip": false,
|
| 14 |
+
"single_word": false
|
| 15 |
+
},
|
| 16 |
+
"pad_token": {
|
| 17 |
+
"content": "[PAD]",
|
| 18 |
+
"lstrip": false,
|
| 19 |
+
"normalized": false,
|
| 20 |
+
"rstrip": false,
|
| 21 |
+
"single_word": false
|
| 22 |
+
},
|
| 23 |
+
"sep_token": {
|
| 24 |
+
"content": "[SEP]",
|
| 25 |
+
"lstrip": false,
|
| 26 |
+
"normalized": false,
|
| 27 |
+
"rstrip": false,
|
| 28 |
+
"single_word": false
|
| 29 |
+
},
|
| 30 |
+
"unk_token": {
|
| 31 |
+
"content": "[UNK]",
|
| 32 |
+
"lstrip": false,
|
| 33 |
+
"normalized": false,
|
| 34 |
+
"rstrip": false,
|
| 35 |
+
"single_word": false
|
| 36 |
+
}
|
| 37 |
+
}
|
checkpoint-235/tokenizer.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-235/tokenizer_config.json
ADDED
|
@@ -0,0 +1,56 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"added_tokens_decoder": {
|
| 3 |
+
"0": {
|
| 4 |
+
"content": "[PAD]",
|
| 5 |
+
"lstrip": false,
|
| 6 |
+
"normalized": false,
|
| 7 |
+
"rstrip": false,
|
| 8 |
+
"single_word": false,
|
| 9 |
+
"special": true
|
| 10 |
+
},
|
| 11 |
+
"100": {
|
| 12 |
+
"content": "[UNK]",
|
| 13 |
+
"lstrip": false,
|
| 14 |
+
"normalized": false,
|
| 15 |
+
"rstrip": false,
|
| 16 |
+
"single_word": false,
|
| 17 |
+
"special": true
|
| 18 |
+
},
|
| 19 |
+
"101": {
|
| 20 |
+
"content": "[CLS]",
|
| 21 |
+
"lstrip": false,
|
| 22 |
+
"normalized": false,
|
| 23 |
+
"rstrip": false,
|
| 24 |
+
"single_word": false,
|
| 25 |
+
"special": true
|
| 26 |
+
},
|
| 27 |
+
"102": {
|
| 28 |
+
"content": "[SEP]",
|
| 29 |
+
"lstrip": false,
|
| 30 |
+
"normalized": false,
|
| 31 |
+
"rstrip": false,
|
| 32 |
+
"single_word": false,
|
| 33 |
+
"special": true
|
| 34 |
+
},
|
| 35 |
+
"103": {
|
| 36 |
+
"content": "[MASK]",
|
| 37 |
+
"lstrip": false,
|
| 38 |
+
"normalized": false,
|
| 39 |
+
"rstrip": false,
|
| 40 |
+
"single_word": false,
|
| 41 |
+
"special": true
|
| 42 |
+
}
|
| 43 |
+
},
|
| 44 |
+
"clean_up_tokenization_spaces": false,
|
| 45 |
+
"cls_token": "[CLS]",
|
| 46 |
+
"do_lower_case": true,
|
| 47 |
+
"extra_special_tokens": {},
|
| 48 |
+
"mask_token": "[MASK]",
|
| 49 |
+
"model_max_length": 512,
|
| 50 |
+
"pad_token": "[PAD]",
|
| 51 |
+
"sep_token": "[SEP]",
|
| 52 |
+
"strip_accents": null,
|
| 53 |
+
"tokenize_chinese_chars": true,
|
| 54 |
+
"tokenizer_class": "BertTokenizer",
|
| 55 |
+
"unk_token": "[UNK]"
|
| 56 |
+
}
|
checkpoint-235/trainer_state.json
ADDED
|
@@ -0,0 +1,121 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"best_global_step": 141,
|
| 3 |
+
"best_metric": 0.8939393939393939,
|
| 4 |
+
"best_model_checkpoint": "./cysecbert-ttp-bert-router/checkpoint-141",
|
| 5 |
+
"epoch": 5.0,
|
| 6 |
+
"eval_steps": 500,
|
| 7 |
+
"global_step": 235,
|
| 8 |
+
"is_hyper_param_search": false,
|
| 9 |
+
"is_local_process_zero": true,
|
| 10 |
+
"is_world_process_zero": true,
|
| 11 |
+
"log_history": [
|
| 12 |
+
{
|
| 13 |
+
"epoch": 1.0,
|
| 14 |
+
"eval_accuracy": 0.8790322580645161,
|
| 15 |
+
"eval_f1": 0.88,
|
| 16 |
+
"eval_loss": 0.23156200349330902,
|
| 17 |
+
"eval_runtime": 0.5457,
|
| 18 |
+
"eval_samples_per_second": 454.439,
|
| 19 |
+
"eval_steps_per_second": 10.994,
|
| 20 |
+
"step": 47
|
| 21 |
+
},
|
| 22 |
+
{
|
| 23 |
+
"epoch": 1.0638297872340425,
|
| 24 |
+
"grad_norm": 244324.0,
|
| 25 |
+
"learning_rate": 1.5829787234042555e-05,
|
| 26 |
+
"loss": 0.393,
|
| 27 |
+
"step": 50
|
| 28 |
+
},
|
| 29 |
+
{
|
| 30 |
+
"epoch": 2.0,
|
| 31 |
+
"eval_accuracy": 0.8870967741935484,
|
| 32 |
+
"eval_f1": 0.889763779527559,
|
| 33 |
+
"eval_loss": 0.24126584827899933,
|
| 34 |
+
"eval_runtime": 0.5406,
|
| 35 |
+
"eval_samples_per_second": 458.709,
|
| 36 |
+
"eval_steps_per_second": 11.098,
|
| 37 |
+
"step": 94
|
| 38 |
+
},
|
| 39 |
+
{
|
| 40 |
+
"epoch": 2.127659574468085,
|
| 41 |
+
"grad_norm": 360389.59375,
|
| 42 |
+
"learning_rate": 1.1574468085106382e-05,
|
| 43 |
+
"loss": 0.2003,
|
| 44 |
+
"step": 100
|
| 45 |
+
},
|
| 46 |
+
{
|
| 47 |
+
"epoch": 3.0,
|
| 48 |
+
"eval_accuracy": 0.8870967741935484,
|
| 49 |
+
"eval_f1": 0.8939393939393939,
|
| 50 |
+
"eval_loss": 0.2832604646682739,
|
| 51 |
+
"eval_runtime": 0.5564,
|
| 52 |
+
"eval_samples_per_second": 445.721,
|
| 53 |
+
"eval_steps_per_second": 10.784,
|
| 54 |
+
"step": 141
|
| 55 |
+
},
|
| 56 |
+
{
|
| 57 |
+
"epoch": 3.1914893617021276,
|
| 58 |
+
"grad_norm": 309358.59375,
|
| 59 |
+
"learning_rate": 7.3191489361702125e-06,
|
| 60 |
+
"loss": 0.1351,
|
| 61 |
+
"step": 150
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"epoch": 4.0,
|
| 65 |
+
"eval_accuracy": 0.8951612903225806,
|
| 66 |
+
"eval_f1": 0.8916666666666667,
|
| 67 |
+
"eval_loss": 0.2920267879962921,
|
| 68 |
+
"eval_runtime": 0.5436,
|
| 69 |
+
"eval_samples_per_second": 456.231,
|
| 70 |
+
"eval_steps_per_second": 11.038,
|
| 71 |
+
"step": 188
|
| 72 |
+
},
|
| 73 |
+
{
|
| 74 |
+
"epoch": 4.25531914893617,
|
| 75 |
+
"grad_norm": 91590.8125,
|
| 76 |
+
"learning_rate": 3.0638297872340428e-06,
|
| 77 |
+
"loss": 0.0876,
|
| 78 |
+
"step": 200
|
| 79 |
+
},
|
| 80 |
+
{
|
| 81 |
+
"epoch": 5.0,
|
| 82 |
+
"eval_accuracy": 0.8911290322580645,
|
| 83 |
+
"eval_f1": 0.889795918367347,
|
| 84 |
+
"eval_loss": 0.3170631229877472,
|
| 85 |
+
"eval_runtime": 0.5567,
|
| 86 |
+
"eval_samples_per_second": 445.496,
|
| 87 |
+
"eval_steps_per_second": 10.778,
|
| 88 |
+
"step": 235
|
| 89 |
+
}
|
| 90 |
+
],
|
| 91 |
+
"logging_steps": 50,
|
| 92 |
+
"max_steps": 235,
|
| 93 |
+
"num_input_tokens_seen": 0,
|
| 94 |
+
"num_train_epochs": 5,
|
| 95 |
+
"save_steps": 500,
|
| 96 |
+
"stateful_callbacks": {
|
| 97 |
+
"EarlyStoppingCallback": {
|
| 98 |
+
"args": {
|
| 99 |
+
"early_stopping_patience": 2,
|
| 100 |
+
"early_stopping_threshold": 0.0
|
| 101 |
+
},
|
| 102 |
+
"attributes": {
|
| 103 |
+
"early_stopping_patience_counter": 2
|
| 104 |
+
}
|
| 105 |
+
},
|
| 106 |
+
"TrainerControl": {
|
| 107 |
+
"args": {
|
| 108 |
+
"should_epoch_stop": false,
|
| 109 |
+
"should_evaluate": false,
|
| 110 |
+
"should_log": false,
|
| 111 |
+
"should_save": true,
|
| 112 |
+
"should_training_stop": true
|
| 113 |
+
},
|
| 114 |
+
"attributes": {}
|
| 115 |
+
}
|
| 116 |
+
},
|
| 117 |
+
"total_flos": 2920532714496000.0,
|
| 118 |
+
"train_batch_size": 48,
|
| 119 |
+
"trial_name": null,
|
| 120 |
+
"trial_params": null
|
| 121 |
+
}
|
checkpoint-235/training_args.bin
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:aa4f68038d34e18bca323801ebee2f977ae0e2c8a6e6a48bf186f9aaacac3db0
|
| 3 |
+
size 5368
|
checkpoint-235/vocab.txt
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
checkpoint-255/config.json
ADDED
|
@@ -0,0 +1,33 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
{
|
| 2 |
+
"architectures": [
|
| 3 |
+
"BertForSequenceClassification"
|
| 4 |
+
],
|
| 5 |
+
"attention_probs_dropout_prob": 0.1,
|
| 6 |
+
"classifier_dropout": null,
|
| 7 |
+
"gradient_checkpointing": false,
|
| 8 |
+
"hidden_act": "gelu",
|
| 9 |
+
"hidden_dropout_prob": 0.1,
|
| 10 |
+
"hidden_size": 768,
|
| 11 |
+
"id2label": {
|
| 12 |
+
"0": "TRAM",
|
| 13 |
+
"1": "ANNOCTR"
|
| 14 |
+
},
|
| 15 |
+
"initializer_range": 0.02,
|
| 16 |
+
"intermediate_size": 3072,
|
| 17 |
+
"label2id": {
|
| 18 |
+
"ANNOCTR": 1,
|
| 19 |
+
"TRAM": 0
|
| 20 |
+
},
|
| 21 |
+
"layer_norm_eps": 1e-12,
|
| 22 |
+
"max_position_embeddings": 512,
|
| 23 |
+
"model_type": "bert",
|
| 24 |
+
"num_attention_heads": 12,
|
| 25 |
+
"num_hidden_layers": 12,
|
| 26 |
+
"pad_token_id": 0,
|
| 27 |
+
"position_embedding_type": "absolute",
|
| 28 |
+
"torch_dtype": "float32",
|
| 29 |
+
"transformers_version": "4.55.2",
|
| 30 |
+
"type_vocab_size": 2,
|
| 31 |
+
"use_cache": true,
|
| 32 |
+
"vocab_size": 30522
|
| 33 |
+
}
|
checkpoint-255/model.safetensors
ADDED
|
@@ -0,0 +1,3 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:7b8c1b24ecd20fdc0656aaa66a2bc6e83aefda5900ce07154dd22df8f4485a25
|
| 3 |
+
size 437958648
|