MaagDeveloper commited on
Commit
fcd61db
·
verified ·
1 Parent(s): 43c8309

Upload folder using huggingface_hub

Browse files
.gitattributes CHANGED
@@ -1,35 +1,10 @@
1
- *.7z filter=lfs diff=lfs merge=lfs -text
2
- *.arrow filter=lfs diff=lfs merge=lfs -text
3
  *.bin filter=lfs diff=lfs merge=lfs -text
4
- *.bz2 filter=lfs diff=lfs merge=lfs -text
5
- *.ckpt filter=lfs diff=lfs merge=lfs -text
6
- *.ftz filter=lfs diff=lfs merge=lfs -text
7
- *.gz filter=lfs diff=lfs merge=lfs -text
8
  *.h5 filter=lfs diff=lfs merge=lfs -text
9
- *.joblib filter=lfs diff=lfs merge=lfs -text
10
- *.lfs.* filter=lfs diff=lfs merge=lfs -text
11
- *.mlmodel filter=lfs diff=lfs merge=lfs -text
12
- *.model filter=lfs diff=lfs merge=lfs -text
13
- *.msgpack filter=lfs diff=lfs merge=lfs -text
14
- *.npy filter=lfs diff=lfs merge=lfs -text
15
- *.npz filter=lfs diff=lfs merge=lfs -text
16
- *.onnx filter=lfs diff=lfs merge=lfs -text
17
- *.ot filter=lfs diff=lfs merge=lfs -text
18
- *.parquet filter=lfs diff=lfs merge=lfs -text
19
- *.pb filter=lfs diff=lfs merge=lfs -text
20
- *.pickle filter=lfs diff=lfs merge=lfs -text
21
- *.pkl filter=lfs diff=lfs merge=lfs -text
22
- *.pt filter=lfs diff=lfs merge=lfs -text
23
- *.pth filter=lfs diff=lfs merge=lfs -text
24
- *.rar filter=lfs diff=lfs merge=lfs -text
25
- *.safetensors filter=lfs diff=lfs merge=lfs -text
26
- saved_model/**/* filter=lfs diff=lfs merge=lfs -text
27
- *.tar.* filter=lfs diff=lfs merge=lfs -text
28
- *.tar filter=lfs diff=lfs merge=lfs -text
29
  *.tflite filter=lfs diff=lfs merge=lfs -text
30
- *.tgz filter=lfs diff=lfs merge=lfs -text
31
- *.wasm filter=lfs diff=lfs merge=lfs -text
32
- *.xz filter=lfs diff=lfs merge=lfs -text
33
- *.zip filter=lfs diff=lfs merge=lfs -text
34
- *.zst filter=lfs diff=lfs merge=lfs -text
35
- *tfevents* filter=lfs diff=lfs merge=lfs -text
 
1
+ *.bin.* filter=lfs diff=lfs merge=lfs -text
2
+ *.lfs.* filter=lfs diff=lfs merge=lfs -text
3
  *.bin filter=lfs diff=lfs merge=lfs -text
 
 
 
 
4
  *.h5 filter=lfs diff=lfs merge=lfs -text
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
5
  *.tflite filter=lfs diff=lfs merge=lfs -text
6
+ *.tar.gz filter=lfs diff=lfs merge=lfs -text
7
+ *.ot filter=lfs diff=lfs merge=lfs -text
8
+ *.onnx filter=lfs diff=lfs merge=lfs -text
9
+ *.msgpack filter=lfs diff=lfs merge=lfs -text
10
+ model.safetensors filter=lfs diff=lfs merge=lfs -text
 
README.md ADDED
@@ -0,0 +1,89 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ ---
2
+ language: ar
3
+ ---
4
+ # Arabic Named Entity Recognition Model
5
+ <a href='https://ko-fi.com/Y8Y71D98TC' target='_blank'><img height='36' style='border:0px;height:36px;' src='https://storage.ko-fi.com/cdn/kofi6.png?v=6' border='0' alt='Buy Me a Coffee at ko-fi.com' /></a>
6
+
7
+ Pretrained BERT-based ([arabic-bert-base](https://huggingface.co/asafaya/bert-base-arabic)) Named Entity Recognition model for Arabic.
8
+
9
+ The pre-trained model can recognize the following entities:
10
+ 1. **PERSON**
11
+
12
+ - و هذا ما نفاه المعاون السياسي للرئيس ***نبيه بري*** ، النائب ***علي حسن خليل***
13
+
14
+ - لكن أوساط ***الحريري*** تعتبر أنه ضحى كثيرا في سبيل البلد
15
+
16
+ - و ستفقد الملكة ***إليزابيث الثانية*** بذلك سيادتها على واحدة من آخر ممالك الكومنولث
17
+
18
+ 2. **ORGANIZATION**
19
+
20
+ - حسب أرقام ***البنك الدولي***
21
+
22
+ - أعلن ***الجيش العراقي***
23
+
24
+ - و نقلت وكالة ***رويترز*** عن ثلاثة دبلوماسيين في ***الاتحاد الأوروبي*** ، أن ***بلجيكا*** و ***إيرلندا*** و ***لوكسمبورغ*** تريد أيضاً مناقشة
25
+
26
+ - ***الحكومة الاتحادية*** و ***حكومة إقليم كردستان***
27
+
28
+ - و هو ما يثير الشكوك حول مشاركة النجم البرتغالي في المباراة المرتقبة أمام ***برشلونة*** الإسباني في
29
+
30
+
31
+ 3. ***LOCATION***
32
+
33
+ - الجديد هو تمكين اللاجئين من “ مغادرة الجزيرة تدريجياً و بهدوء إلى ***أثينا*** ”
34
+
35
+ - ***جزيرة ساكيز*** تبعد 1 كم عن ***إزمير***
36
+
37
+
38
+ 4. **DATE**
39
+
40
+ - ***غدا الجمعة***
41
+
42
+ - ***06 أكتوبر 2020***
43
+
44
+ - ***العام السابق***
45
+
46
+
47
+ 5. **PRODUCT**
48
+
49
+ - عبر حسابه ب ***تطبيق “ إنستغرام ”***
50
+
51
+ - الجيل الثاني من ***نظارة الواقع الافتراضي أوكولوس كويست*** تحت اسم " ***أوكولوس كويست 2*** "
52
+
53
+
54
+ 6. **COMPETITION**
55
+
56
+ - عدم المشاركة في ***بطولة فرنسا المفتوحة للتنس***
57
+
58
+ - في مباراة ***كأس السوبر الأوروبي***
59
+
60
+ 7. **PRIZE**
61
+
62
+ - ***جائزة نوبل ل لآداب***
63
+
64
+ - الذي فاز ب ***جائزة “ إيمي ” لأفضل دور مساند***
65
+
66
+ 8. **EVENT**
67
+
68
+ - تسجّل أغنية جديدة خاصة ب ***العيد الوطني السعودي***
69
+
70
+ - ***مهرجان المرأة يافوية*** في دورته الرابعة
71
+
72
+ 9. **DISEASE**
73
+
74
+ - في مكافحة فيروس ***كورونا*** و عدد من الأمراض
75
+
76
+ - الأزمات المشابهة مثل “ ***انفلونزا الطيور*** ” و ” ***انفلونزا الخنازير***
77
+
78
+ ## Example
79
+
80
+ [Find here a complete example to use this model](https://github.com/hatmimoha/arabic-ner)
81
+
82
+ ## Training Corpus
83
+
84
+ The training corpus is made of 378.000 tokens (14.000 sentences) collected from the Web and annotated manually.
85
+
86
+
87
+ ## Results
88
+
89
+ The results on a valid corpus made of 30.000 tokens shows an F-measure of ~87%.
config.json ADDED
@@ -0,0 +1,68 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "_name_or_path": "asafaya/bert-base-arabic",
3
+ "architectures": [
4
+ "BertForTokenClassification"
5
+ ],
6
+ "attention_probs_dropout_prob": 0.1,
7
+ "classifier_dropout": null,
8
+ "gradient_checkpointing": false,
9
+ "hidden_act": "gelu",
10
+ "hidden_dropout_prob": 0.1,
11
+ "hidden_size": 768,
12
+ "id2label": {
13
+ "0": "B-COMPETITION",
14
+ "1": "B-DATE",
15
+ "2": "B-DISEASE",
16
+ "3": "B-EVENT",
17
+ "4": "B-LOCATION",
18
+ "5": "B-ORGANIZATION",
19
+ "6": "B-PERSON",
20
+ "7": "B-PRICE",
21
+ "8": "B-PRODUCT",
22
+ "9": "I-COMPETITION",
23
+ "10": "I-DATE",
24
+ "11": "I-DISEASE",
25
+ "12": "I-EVENT",
26
+ "13": "I-LOCATION",
27
+ "14": "I-ORGANIZATION",
28
+ "15": "I-PERSON",
29
+ "16": "I-PRICE",
30
+ "17": "I-PRODUCT",
31
+ "18": "O"
32
+ },
33
+ "initializer_range": 0.02,
34
+ "intermediate_size": 3072,
35
+ "label2id": {
36
+ "B-COMPETITION": 0,
37
+ "B-DATE": 1,
38
+ "B-DISEASE": 2,
39
+ "B-EVENT": 3,
40
+ "B-LOCATION": 4,
41
+ "B-ORGANIZATION": 5,
42
+ "B-PERSON": 6,
43
+ "B-PRICE": 7,
44
+ "B-PRODUCT": 8,
45
+ "I-COMPETITION": 9,
46
+ "I-DATE": 10,
47
+ "I-DISEASE": 11,
48
+ "I-EVENT": 12,
49
+ "I-LOCATION": 13,
50
+ "I-ORGANIZATION": 14,
51
+ "I-PERSON": 15,
52
+ "I-PRICE": 16,
53
+ "I-PRODUCT": 17,
54
+ "O": 18
55
+ },
56
+ "layer_norm_eps": 1e-12,
57
+ "max_position_embeddings": 512,
58
+ "model_type": "bert",
59
+ "num_attention_heads": 12,
60
+ "num_hidden_layers": 12,
61
+ "output_past": true,
62
+ "pad_token_id": 0,
63
+ "position_embedding_type": "absolute",
64
+ "torch_dtype": "float32",
65
+ "transformers_version": "4.24.0",
66
+ "type_vocab_size": 2,
67
+ "vocab_size": 32000
68
+ }
flax_model.msgpack ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:a744745bcb0b1427238607d692de80c881cf6362517191bcc1f36ad8e5aafad0
3
+ size 440172594
model.safetensors ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3764c2575a6db5ed5efb4ff633f5067c137243e997bff6b3037811299594b70d
3
+ size 440192992
model_args.json ADDED
@@ -0,0 +1 @@
 
 
1
+ {"adam_epsilon": 1e-08, "best_model_dir": "outputs/best_model", "cache_dir": "cache_dir/", "config": {}, "custom_layer_parameters": [], "custom_parameter_groups": [], "dataloader_num_workers": 1, "do_lower_case": false, "dynamic_quantize": false, "early_stopping_consider_epochs": false, "early_stopping_delta": 0, "early_stopping_metric": "eval_loss", "early_stopping_metric_minimize": true, "early_stopping_patience": 3, "encoding": null, "eval_batch_size": 8, "evaluate_during_training": false, "evaluate_during_training_silent": true, "evaluate_during_training_steps": 2000, "evaluate_during_training_verbose": false, "evaluate_each_epoch": true, "fp16": true, "gradient_accumulation_steps": 8, "learning_rate": 4e-05, "local_rank": -1, "logging_steps": 50, "manual_seed": null, "max_grad_norm": 1.0, "max_seq_length": 256, "model_name": "/content/drive/My Drive/Colab Notebooks/output-last/", "model_type": "bert", "multiprocessing_chunksize": 500, "n_gpu": 1, "no_cache": false, "no_save": false, "num_train_epochs": 3, "output_dir": "/content/drive/My Drive/Colab Notebooks/output/", "overwrite_output_dir": true, "process_count": 1, "quantized_model": false, "reprocess_input_data": true, "save_best_model": true, "save_eval_checkpoints": true, "save_model_every_epoch": false, "save_optimizer_and_scheduler": true, "save_steps": -1, "silent": false, "tensorboard_dir": null, "thread_count": null, "train_batch_size": 10, "train_custom_parameters_only": false, "use_cached_eval_features": false, "use_early_stopping": false, "use_multiprocessing": true, "wandb_kwargs": {}, "wandb_project": null, "warmup_ratio": 0.06, "warmup_steps": 266, "weight_decay": 0, "skip_special_tokens": true, "model_class": "NERModel", "classification_report": false, "labels_list": ["B-PERSON", "I-PERSON", "B-ORGANIZATION", "I-ORGANIZATION", "B-LOCATION", "I-LOCATION", "B-DATE", "I-DATE", "B-COMPETITION", "I-COMPETITION", "B-PRICE", "I-PRICE", "O", "B-PRODUCT", "I-PRODUCT", "B-EVENT", "I-EVENT", "B-DISEASE", "I-DISEASE"], "lazy_loading": false, "lazy_loading_start_line": 0, "onnx": false}
pytorch_model.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:3a70dc2fa360ac2997727c20ee7a91e028595f2176238cac909ab79c9e5e3e29
3
+ size 440235889
special_tokens_map.json ADDED
@@ -0,0 +1,7 @@
 
 
 
 
 
 
 
 
1
+ {
2
+ "cls_token": "[CLS]",
3
+ "mask_token": "[MASK]",
4
+ "pad_token": "[PAD]",
5
+ "sep_token": "[SEP]",
6
+ "unk_token": "[UNK]"
7
+ }
tf_model.h5 ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:eb0c57bbabb33b3e549d4facf6e03ab09810d3a00553ee68e29e81a0447565ae
3
+ size 442815900
tokenizer.json ADDED
The diff for this file is too large to render. See raw diff
 
tokenizer_config.json ADDED
@@ -0,0 +1,16 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ {
2
+ "cls_token": "[CLS]",
3
+ "do_basic_tokenize": true,
4
+ "do_lower_case": true,
5
+ "full_tokenizer_file": null,
6
+ "mask_token": "[MASK]",
7
+ "name_or_path": "asafaya/bert-base-arabic",
8
+ "never_split": null,
9
+ "pad_token": "[PAD]",
10
+ "sep_token": "[SEP]",
11
+ "special_tokens_map_file": null,
12
+ "strip_accents": null,
13
+ "tokenize_chinese_chars": true,
14
+ "tokenizer_class": "BertTokenizer",
15
+ "unk_token": "[UNK]"
16
+ }
training_args.bin ADDED
@@ -0,0 +1,3 @@
 
 
 
 
1
+ version https://git-lfs.github.com/spec/v1
2
+ oid sha256:4b416605fdb4edc9a7a01f7e76cb8c86bf6f01e5f051c63f336895ac1adaca65
3
+ size 3375
vocab.txt ADDED
The diff for this file is too large to render. See raw diff