Training in progress, epoch 1

Browse files

Files changed (13) hide show

README.md +109 -0
adapter_config.json +33 -0
adapter_model.safetensors +3 -0
added_tokens.json +4 -0
all_results.json +13 -0
eval_results.json +8 -0
merges.txt +0 -0
special_tokens_map.json +28 -0
tokenizer.json +0 -0
tokenizer_config.json +41 -0
train_results.json +8 -0
training_args.bin +3 -0
vocab.json +0 -0

README.md ADDED Viewed

	@@ -0,0 +1,109 @@

+---
+library_name: peft
+license: mit
+base_model: gpt2
+tags:
+- generated_from_trainer
+model-index:
+- name: Se124M10KInfKeyValue
+  results: []
+---
+<!-- This model card has been generated automatically according to the information the Trainer had access to. You
+should probably proofread and complete it, then remove this comment. -->
+# Se124M10KInfKeyValue
+This model is a fine-tuned version of [gpt2](https://huggingface.co/gpt2) on an unknown dataset.
+It achieves the following results on the evaluation set:
+- Loss: 0.5416
+## Model description
+More information needed
+## Intended uses & limitations
+More information needed
+## Training and evaluation data
+More information needed
+## Training procedure
+### Training hyperparameters
+The following hyperparameters were used during training:
+- learning_rate: 5e-05
+- train_batch_size: 32
+- eval_batch_size: 32
+- seed: 42
+- optimizer: Use OptimizerNames.ADAMW_TORCH with betas=(0.9,0.999) and epsilon=1e-08 and optimizer_args=No additional optimizer arguments
+- lr_scheduler_type: linear
+- num_epochs: 50
+- mixed_precision_training: Native AMP
+### Training results
+| Training Loss | Epoch | Step  | Validation Loss |
+|:-------------:|:-----:|:-----:|:---------------:|
+| 0.3613        | 1.0   | 225   | 0.9669          |
+| 0.2223        | 2.0   | 450   | 0.7171          |
+| 0.1879        | 3.0   | 675   | 0.6545          |
+| 0.1715        | 4.0   | 900   | 0.6227          |
+| 0.1658        | 5.0   | 1125  | 0.6118          |
+| 0.1606        | 6.0   | 1350  | 0.6041          |
+| 0.1571        | 7.0   | 1575  | 0.5922          |
+| 0.1547        | 8.0   | 1800  | 0.5869          |
+| 0.1541        | 9.0   | 2025  | 0.5817          |
+| 0.1499        | 10.0  | 2250  | 0.5814          |
+| 0.1511        | 11.0  | 2475  | 0.5738          |
+| 0.1479        | 12.0  | 2700  | 0.5730          |
+| 0.1487        | 13.0  | 2925  | 0.5697          |
+| 0.1449        | 14.0  | 3150  | 0.5665          |
+| 0.1448        | 15.0  | 3375  | 0.5653          |
+| 0.1435        | 16.0  | 3600  | 0.5645          |
+| 0.1455        | 17.0  | 3825  | 0.5612          |
+| 0.1437        | 18.0  | 4050  | 0.5586          |
+| 0.1417        | 19.0  | 4275  | 0.5565          |
+| 0.1434        | 20.0  | 4500  | 0.5580          |
+| 0.1439        | 21.0  | 4725  | 0.5561          |
+| 0.1421        | 22.0  | 4950  | 0.5552          |
+| 0.1414        | 23.0  | 5175  | 0.5532          |
+| 0.1396        | 24.0  | 5400  | 0.5505          |
+| 0.1392        | 25.0  | 5625  | 0.5521          |
+| 0.1413        | 26.0  | 5850  | 0.5517          |
+| 0.1385        | 27.0  | 6075  | 0.5478          |
+| 0.1413        | 28.0  | 6300  | 0.5485          |
+| 0.1411        | 29.0  | 6525  | 0.5500          |
+| 0.139         | 30.0  | 6750  | 0.5482          |
+| 0.1402        | 31.0  | 6975  | 0.5480          |
+| 0.1397        | 32.0  | 7200  | 0.5462          |
+| 0.1402        | 33.0  | 7425  | 0.5448          |
+| 0.1405        | 34.0  | 7650  | 0.5472          |
+| 0.1374        | 35.0  | 7875  | 0.5437          |
+| 0.1373        | 36.0  | 8100  | 0.5446          |
+| 0.1386        | 37.0  | 8325  | 0.5439          |
+| 0.1372        | 38.0  | 8550  | 0.5438          |
+| 0.1383        | 39.0  | 8775  | 0.5431          |
+| 0.1375        | 40.0  | 9000  | 0.5428          |
+| 0.1398        | 41.0  | 9225  | 0.5431          |
+| 0.1394        | 42.0  | 9450  | 0.5418          |
+| 0.1395        | 43.0  | 9675  | 0.5423          |
+| 0.1377        | 44.0  | 9900  | 0.5423          |
+| 0.1366        | 45.0  | 10125 | 0.5422          |
+| 0.138         | 46.0  | 10350 | 0.5419          |
+| 0.136         | 47.0  | 10575 | 0.5416          |
+| 0.138         | 48.0  | 10800 | 0.5418          |
+| 0.1373        | 49.0  | 11025 | 0.5417          |
+| 0.1365        | 50.0  | 11250 | 0.5416          |
+### Framework versions
+- PEFT 0.15.1
+- Transformers 4.51.3
+- Pytorch 2.6.0+cu118
+- Datasets 3.5.0
+- Tokenizers 0.21.1

adapter_config.json ADDED Viewed

	@@ -0,0 +1,33 @@

+{
+  "alpha_pattern": {},
+  "auto_mapping": null,
+  "base_model_name_or_path": "gpt2",
+  "bias": "none",
+  "corda_config": null,
+  "eva_config": null,
+  "exclude_modules": null,
+  "fan_in_fan_out": true,
+  "inference_mode": true,
+  "init_lora_weights": true,
+  "layer_replication": null,
+  "layers_pattern": null,
+  "layers_to_transform": null,
+  "loftq_config": {},
+  "lora_alpha": 32,
+  "lora_bias": false,
+  "lora_dropout": 0.05,
+  "megatron_config": null,
+  "megatron_core": "megatron.core",
+  "modules_to_save": null,
+  "peft_type": "LORA",
+  "r": 8,
+  "rank_pattern": {},
+  "revision": null,
+  "target_modules": [
+    "c_attn"
+  ],
+  "task_type": "CAUSAL_LM",
+  "trainable_token_indices": null,
+  "use_dora": false,
+  "use_rslora": false
+}

adapter_model.safetensors ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:139d9024b51698d994da93deb5d5081e6a5fb96c69ab468a0dfea7cb39887446
+size 309974336

added_tokens.json ADDED Viewed

	@@ -0,0 +1,4 @@

+{
+  "<|endofex|>": 50258,
+  "<|startofex|>": 50257
+}

all_results.json ADDED Viewed

	@@ -0,0 +1,13 @@

+{
+    "epoch": 50.0,
+    "eval_loss": 0.5416154265403748,
+    "eval_runtime": 9.3336,
+    "eval_samples_per_second": 160.924,
+    "eval_steps_per_second": 5.036,
+    "perplexity": 1.7187811854679051,
+    "total_flos": 2.35191607492608e+16,
+    "train_loss": 0.15202610808478462,
+    "train_runtime": 1233.6469,
+    "train_samples_per_second": 290.845,
+    "train_steps_per_second": 9.119
+}

eval_results.json ADDED Viewed

	@@ -0,0 +1,8 @@

+{
+    "epoch": 50.0,
+    "eval_loss": 0.5416154265403748,
+    "eval_runtime": 9.3336,
+    "eval_samples_per_second": 160.924,
+    "eval_steps_per_second": 5.036,
+    "perplexity": 1.7187811854679051
+}

merges.txt ADDED Viewed

The diff for this file is too large to render. See raw diff

special_tokens_map.json ADDED Viewed

	@@ -0,0 +1,28 @@

+{
+  "additional_special_tokens": [
+    {
+      "content": "<|startofex|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    },
+    {
+      "content": "<|endofex|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false
+    }
+  ],
+  "bos_token": "<|endoftext|>",
+  "eos_token": "<|endoftext|>",
+  "pad_token": {
+    "content": "<|endoftext|>",
+    "lstrip": false,
+    "normalized": false,
+    "rstrip": false,
+    "single_word": false
+  },
+  "unk_token": "<|endoftext|>"
+}

tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

tokenizer_config.json ADDED Viewed

	@@ -0,0 +1,41 @@

+{
+  "add_prefix_space": false,
+  "added_tokens_decoder": {
+    "50256": {
+      "content": "<|endoftext|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "50257": {
+      "content": "<|startofex|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    },
+    "50258": {
+      "content": "<|endofex|>",
+      "lstrip": false,
+      "normalized": false,
+      "rstrip": false,
+      "single_word": false,
+      "special": true
+    }
+  },
+  "additional_special_tokens": [
+    "<|startofex|>",
+    "<|endofex|>"
+  ],
+  "bos_token": "<|endoftext|>",
+  "clean_up_tokenization_spaces": false,
+  "eos_token": "<|endoftext|>",
+  "extra_special_tokens": {},
+  "model_max_length": 1024,
+  "pad_token": "<|endoftext|>",
+  "tokenizer_class": "GPT2Tokenizer",
+  "unk_token": "<|endoftext|>"
+}

train_results.json ADDED Viewed

	@@ -0,0 +1,8 @@

+{
+    "epoch": 50.0,
+    "total_flos": 2.35191607492608e+16,
+    "train_loss": 0.15202610808478462,
+    "train_runtime": 1233.6469,
+    "train_samples_per_second": 290.845,
+    "train_steps_per_second": 9.119
+}

training_args.bin ADDED Viewed

	@@ -0,0 +1,3 @@

+version https://git-lfs.github.com/spec/v1
+oid sha256:e52ca09d021654889cbee3a5e3895e31b50a1e0fc58d8668ef20b49a339db24f
+size 5368

vocab.json ADDED Viewed

The diff for this file is too large to render. See raw diff