Upload folder using huggingface_hub

Files changed (9) hide show

checkpoint-33/config.json ADDED Viewed

+{
+  "architectures": [
+    "LlamaForCausalLM"
+  ],
+  "attention_bias": false,
+  "attention_dropout": 0.0,
+  "bos_token_id": 1,
+  "dtype": "float32",
+  "eos_token_id": 2,
+  "head_dim": 64,
+  "hidden_act": "silu",
+  "hidden_size": 1280,
+  "initializer_range": 0.02,
+  "intermediate_size": 3584,
+  "max_position_embeddings": 1024,
+  "mlp_bias": false,
+  "model_type": "llama",
+  "num_attention_heads": 20,
+  "num_hidden_layers": 20,
+  "num_key_value_heads": 5,
+  "pad_token_id": null,
+  "pretraining_tp": 1,
+  "rms_norm_eps": 1e-06,
+  "rope_parameters": {
+    "rope_theta": 10000.0,
+    "rope_type": "default"
+  },
+  "tie_word_embeddings": true,
+  "transformers_version": "5.0.0",
+  "use_cache": false,
+  "vocab_size": 32000
+}

checkpoint-33/generation_config.json ADDED Viewed

+{
+  "_from_model_config": true,
+  "bos_token_id": 1,
+  "eos_token_id": 2,
+  "output_attentions": false,
+  "output_hidden_states": false,
+  "transformers_version": "5.0.0",
+  "use_cache": true
+}

checkpoint-33/optimizer.pt ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:329ef27ea6e343bf111db0f0253632ff9c00cd65ded1da92acd086fff9f3b900
+size 1053800587

checkpoint-33/rng_state.pth ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:61c19bab1174704a4a4441475683bf1270277af15d2e2c95e964789128e482c4
+size 14645

checkpoint-33/scheduler.pt ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:f12f17f5a160b07c5f56feecfde50f733fff12c335ec3258ac3ba0d121e96f74
+size 1465

checkpoint-33/tokenizer.json ADDED Viewed

The diff for this file is too large to render. See raw diff

checkpoint-33/tokenizer_config.json ADDED Viewed

+{
+  "backend": "tokenizers",
+  "bos_token": "<bos>",
+  "eos_token": "<|im_end|>",
+  "extra_special_tokens": [
+    "<|im_start|>",
+    "<eos>",
+    "<|nl|>",
+    "<|nl2|>"
+  ],
+  "model_max_length": 1000000000000000019884624838656,
+  "pad_token": "<pad>",
+  "tokenizer_class": "TokenizersBackend",
+  "unk_token": "<unk>"
+}

checkpoint-33/trainer_state.json ADDED Viewed

+{
+  "best_global_step": null,
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 1.0,
+  "eval_steps": 500,
+  "global_step": 33,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.6189555125725339,
+      "grad_norm": 0.8591008186340332,
+      "learning_rate": 8.75e-05,
+      "loss": 8.850434875488281,
+      "step": 20
+    }
+  ],
+  "logging_steps": 20,
+  "max_steps": 33,
+  "num_input_tokens_seen": 0,
+  "num_train_epochs": 1,
+  "save_steps": 250,
+  "stateful_callbacks": {
+    "TrainerControl": {
+      "args": {
+        "should_epoch_stop": false,
+        "should_evaluate": false,
+        "should_log": false,
+        "should_save": true,
+        "should_training_stop": true
+      },
+      "attributes": {}
+    }
+  },
+  "total_flos": 2269404887777280.0,
+  "train_batch_size": 1,
+  "trial_name": null,
+  "trial_params": null
+}

checkpoint-33/training_args.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:006ba34976e96eb42331e7bbc6d86ea0e690d890bbbaa95ac7ec7010a446cb20
+size 5201