Training in progress, step 1000

Files changed (14) hide show

.gitignore ADDED Viewed

	@@ -0,0 +1 @@


1	+ checkpoint-*/

config.json ADDED Viewed

+{
+  "_name_or_path": "/content/Socrat_4",
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.1,
+  "bos_token_id": 50256,
+  "embd_pdrop": 0.1,
+  "eos_token_id": 50256,
+  "id2label": {
+    "0": "LABEL_0"
+  },
+  "initializer_range": 0.02,
+  "label2id": {
+    "LABEL_0": 0
+  },
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_ctx": 2048,
+  "n_embd": 1024,
+  "n_head": 16,
+  "n_inner": null,
+  "n_layer": 24,
+  "n_positions": 2048,
+  "n_special": 0,
+  "output_past": true,
+  "predict_special_tokens": true,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.1,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.27.2",
+  "use_cache": true,
+  "vocab_size": 50257
+}

last-checkpoint/config.json ADDED Viewed

+{
+  "_name_or_path": "/content/Socrat_4",
+  "activation_function": "gelu_new",
+  "architectures": [
+    "GPT2LMHeadModel"
+  ],
+  "attn_pdrop": 0.1,
+  "bos_token_id": 50256,
+  "embd_pdrop": 0.1,
+  "eos_token_id": 50256,
+  "id2label": {
+    "0": "LABEL_0"
+  },
+  "initializer_range": 0.02,
+  "label2id": {
+    "LABEL_0": 0
+  },
+  "layer_norm_epsilon": 1e-05,
+  "model_type": "gpt2",
+  "n_ctx": 2048,
+  "n_embd": 1024,
+  "n_head": 16,
+  "n_inner": null,
+  "n_layer": 24,
+  "n_positions": 2048,
+  "n_special": 0,
+  "output_past": true,
+  "predict_special_tokens": true,
+  "reorder_and_upcast_attn": false,
+  "resid_pdrop": 0.1,
+  "scale_attn_by_inverse_layer_idx": false,
+  "scale_attn_weights": true,
+  "summary_activation": null,
+  "summary_first_dropout": 0.1,
+  "summary_proj_to_labels": true,
+  "summary_type": "cls_index",
+  "summary_use_proj": true,
+  "torch_dtype": "float32",
+  "transformers_version": "4.27.2",
+  "use_cache": true,
+  "vocab_size": 50257
+}

last-checkpoint/generation_config.json ADDED Viewed

+{
+  "_from_model_config": true,
+  "bos_token_id": 50256,
+  "eos_token_id": 50256,
+  "transformers_version": "4.27.2"
+}

last-checkpoint/optimizer.pt ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:d951dcccb3138caeed1ab4351056967ab2bf699172236aa45edc1328eeb5dfcb
+size 2847145157

last-checkpoint/pytorch_model.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:a75bb415d99c0da358739989a91937fe22b77d82618cb1946977a2b4e860ef3b
+size 1524261149

last-checkpoint/rng_state.pth ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:478d2a12e14dda206ad13c93b28e9383719e71464f75e206b9b3b5a4f9e3ffb4
+size 14575

last-checkpoint/scheduler.pt ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:6fd63048cac0dc24c318a6baa96db9ac8e3f1ebf138acdbe8249be7a05bddc22
+size 627

last-checkpoint/trainer_state.json ADDED Viewed

+{
+  "best_metric": null,
+  "best_model_checkpoint": null,
+  "epoch": 0.25012506253126565,
+  "global_step": 1000,
+  "is_hyper_param_search": false,
+  "is_local_process_zero": true,
+  "is_world_process_zero": true,
+  "log_history": [
+    {
+      "epoch": 0.13,
+      "learning_rate": 1.949974987493747e-05,
+      "loss": 2.8007,
+      "step": 500
+    },
+    {
+      "epoch": 0.25,
+      "learning_rate": 1.8999499749874938e-05,
+      "loss": 2.7961,
+      "step": 1000
+    },
+    {
+      "epoch": 0.25,
+      "eval_loss": 3.1374855041503906,
+      "eval_runtime": 136.8005,
+      "eval_samples_per_second": 15.475,
+      "eval_steps_per_second": 5.161,
+      "step": 1000
+    }
+  ],
+  "max_steps": 19990,
+  "num_train_epochs": 5,
+  "total_flos": 1414817464320000.0,
+  "trial_name": null,
+  "trial_params": null
+}

last-checkpoint/training_args.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:a27a26c548ae69b704f6121b2840b0eae566d932411df704e8d8478fc6c30e2e
+size 3579

pytorch_model.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:a75bb415d99c0da358739989a91937fe22b77d82618cb1946977a2b4e860ef3b
+size 1524261149

runs/Mar21_09-06-28_da74753029b6/1679389598.649497/events.out.tfevents.1679389598.da74753029b6.214.1 ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:b828ef7e07101367a9bc7e2aaeda6d7cda41c543dfed94de02b8c9187534d920
+size 5742

runs/Mar21_09-06-28_da74753029b6/events.out.tfevents.1679389598.da74753029b6.214.0 ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:6112b973c976203b085883ed9cf44eb7898dc00e0d719b56ed9eac6c7cac2a96
+size 4783

training_args.bin ADDED Viewed

+version https://git-lfs.github.com/spec/v1
+oid sha256:a27a26c548ae69b704f6121b2840b0eae566d932411df704e8d8478fc6c30e2e
+size 3579