BharatVLM commited on Jun 5, 2025

Commit

81472f1

verified ·

1 Parent(s): 3d930be

Upload folder using huggingface_hub

Browse files

Files changed (18) hide show

README.md +83 -3
all_results.json +16 -0
config.json +35 -0
eval_results.json +10 -0
generation_config.json +7 -0
merges.txt +0 -0
model.safetensors +3 -0
runs/May13_18-16-24_uma/events.out.tfevents.1747140395.uma.2750189.0 +3 -0
runs/May13_18-16-24_uma/events.out.tfevents.1747231217.uma.2750189.1 +3 -0
runs/May14_22-41-05_uma/events.out.tfevents.1747242678.uma.3260955.0 +3 -0
runs/May14_22-41-05_uma/events.out.tfevents.1747317807.uma.3260955.1 +3 -0
special_tokens_map.json +37 -0
tokenizer.json +0 -0
tokenizer_config.json +56 -0
train_results.json +9 -0
trainer_state.json +0 -0
training_args.bin +3 -0
vocab.json +0 -0

README.md CHANGED Viewed

@@ -1,3 +1,83 @@
----
-license: cc-by-nc-4.0
----

+---
+library_name: transformers
+tags:
+- gpt2
+- assamese
+- language-model
+- text-generation
+- low-resource
+- educational
+- research
+- generated_from_trainer
+metrics:
+- accuracy
+model-index:
+- name: Assamese GPT-2
+  results: []
+---
+# Assamese GPT-2 Model
+This is a GPT-2 language model trained from scratch on Assamese monolingual text, using data from **IndicCorpV2** and **OSCAR**. The model is developed for **educational and research purposes** to support natural language understanding and generation tasks in Assamese — a low-resource language.
+## 📖 Model Description
+The Assamese GPT-2 model is based on the standard GPT-2 decoder-only transformer architecture. It is capable of generating grammatically coherent and contextually relevant Assamese text and serves as a foundation for downstream NLP tasks such as:
+- Language modeling
+- Text completion/generation
+- Fine-tuning for classification or summarization
+## ✅ Intended Uses
+- Academic research on Assamese NLP
+- Training and benchmarking in educational settings
+- Exploration of low-resource language modeling
+## 🚫 Limitations
+- Trained on general-domain monolingual data, may not perform well on domain-specific texts (e.g., legal, medical).
+- Might generate biased, incomplete, or hallucinated outputs.
+- Not suitable for production use or deployment in sensitive applications.
+## 📚 Training and Evaluation Data
+The model was trained using Assamese monolingual data collected from:
+- **IndicCorpV2**: A curated collection of web-crawled and processed data for Indic languages.
+- **OSCAR (Open Super-large Crawled ALMAnaCH coRpus)**: Filtered web-crawled corpus available through Hugging Face datasets.
+Data preprocessing included:
+- Unicode normalization
+- Removal of noisy characters and malformed tokens
+- Sentence segmentation using Assamese-specific heuristics
+## 🧪 Training Procedure
+### Hyperparameters
+- Learning rate: 5e-5
+- Epochs: 20
+- Batch size: 64
+- Optimizer: AdamW (β₁=0.9, β₂=0.999, ε=1e-8)
+- Scheduler: Linear
+- Mixed Precision: Native AMP
+- Seed: 42
+### Results
+- Final Evaluation Loss: -29.1890
+- Accuracy: 0.3452
+## 🚀 Example Usage
+```python
+from transformers import GPT2LMHeadModel, GPT2Tokenizer
+model = GPT2LMHeadModel.from_pretrained("your-username/gpt2_assamese_model")
+tokenizer = GPT2Tokenizer.from_pretrained("your-username/gpt2_assamese_model")
+prompt = "অসমৰ ইতিহাস"
+inputs = tokenizer(prompt, return_tensors="pt")
+outputs = model.generate(**inputs, max_length=50, do_sample=True)
+print(tokenizer.decode(outputs[0], skip_special_tokens=True))
+```

all_results.json ADDED Viewed

	@@ -0,0 +1,16 @@

+{
+    "epoch": 20.0,
+    "eval_accuracy": 0.3452154413455673,
+    "eval_loss": -29.189016342163086,
+    "eval_runtime": 2434.4538,
+    "eval_samples": 10618,
+    "eval_samples_per_second": 4.362,
+    "eval_steps_per_second": 0.068,
+    "perplexity": 2.1055776904663675e-13,
+    "total_flos": 2.10597197119488e+18,
+    "train_loss": 0.33622724912249125,
+    "train_runtime": 72693.7372,
+    "train_samples": 201496,
+    "train_samples_per_second": 55.437,
+    "train_steps_per_second": 0.866
+}

config.json ADDED Viewed

	@@ -0,0 +1,35 @@

+{
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.1,
+  "bos_token_id": 0,
+  "embd_pdrop": 0.1,
+  "eos_token_id": 2,
+  "initializer_range": 0.02,
+  "layer_norm_epsilon": 1e-05,
+  "mask_token_id": 4,
+  "model_type": "gpt2",
+  "n_ctx": 1024,
+  "n_embd": 768,
+  "n_head": 12,
+  "n_inner": null,
+  "n_layer": 12,
+  "n_positions": 1024,
+  "pad_token_id": 1,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.1,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.52.0.dev0",
+  "unk_token_id": 3,
+  "use_cache": true,
+  "vocab_size": 50000
+}

eval_results.json ADDED Viewed

	@@ -0,0 +1,10 @@

+{
+    "epoch": 20.0,
+    "eval_accuracy": 0.3452154413455673,
+    "eval_loss": -29.189016342163086,
+    "eval_runtime": 2434.4538,
+    "eval_samples": 10618,
+    "eval_samples_per_second": 4.362,
+    "eval_steps_per_second": 0.068,
+    "perplexity": 2.1055776904663675e-13
+}

generation_config.json ADDED Viewed

	@@ -0,0 +1,7 @@

+{
+  "_from_model_config": true,
+  "bos_token_id": 0,
+  "eos_token_id": 2,
+  "pad_token_id": 1,
+  "transformers_version": "4.52.0.dev0"
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d9ebdd0b85e0ddcbbb57a300bc90ad61342ac8759e4fbdc264220f6fdadc66ea
+size 496984704

runs/May13_18-16-24_uma/events.out.tfevents.1747140395.uma.2750189.0 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:d266216ac4d0715e394e0ce4da673f0d4b664b55e16b640ec2d9aa4cb260117c
+size 139307

runs/May13_18-16-24_uma/events.out.tfevents.1747231217.uma.2750189.1 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:ed35378faa2a97f78db4779c9e82d6a734135b9e5babfc6f68318691862eeb90
+size 417

runs/May14_22-41-05_uma/events.out.tfevents.1747242678.uma.3260955.0 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:991b9019cbd4fccf2b340e5663fae8dd86f4d2b02aefd1a5be9ef37b4f837b81
+size 140858

runs/May14_22-41-05_uma/events.out.tfevents.1747317807.uma.3260955.1 ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:615c02f60eb6d13bcb05b36283e7ac74f556d4967a1a2bc2a1bbc70abfcd57c2
+size 417

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,37 @@

+{
+  "bos_token": {
+    "content": "<s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "eos_token": {
+    "content": "</s>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "mask_token": {
+    "content": "<mask>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "pad_token": {
+    "content": "<pad>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": {
+    "content": "<unk>",
+    "lstrip": false,
+    "normalized": true,
+    "rstrip": false,
+    "single_word": false
+  }
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,56 @@

+{
+  "add_bos_token": false,
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "0": {
+      "content": "<s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "1": {
+      "content": "<pad>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "2": {
+      "content": "</s>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "3": {
+      "content": "<unk>",
+      "lstrip": false,
+      "normalized": true,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "4": {
+      "content": "<mask>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "bos_token": "<s>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "</s>",
+  "errors": "replace",
+  "extra_special_tokens": {},
+  "mask_token": "<mask>",
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<pad>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<unk>"
+}

train_results.json ADDED Viewed

	@@ -0,0 +1,9 @@

+{
+    "epoch": 20.0,
+    "total_flos": 2.10597197119488e+18,
+    "train_loss": 0.33622724912249125,
+    "train_runtime": 72693.7372,
+    "train_samples": 201496,
+    "train_samples_per_second": 55.437,
+    "train_steps_per_second": 0.866
+}

trainer_state.json ADDED Viewed

The diff for this file is too large to render. See raw diff

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:6c0474f633b98bb6445345154f9ee9b4af4bbac2841b138afd5cbfc3c3f825b4
+size 5304

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff