Instructions to use raqibcodes/c2c-checkpoints-v2 with libraries, inference providers, notebooks, and local apps. Follow these links to get started.
- Libraries
- Transformers
How to use raqibcodes/c2c-checkpoints-v2 with Transformers:
# Load model directly from transformers import AutoModel model = AutoModel.from_pretrained("raqibcodes/c2c-checkpoints-v2", device_map="auto") - Notebooks
- Google Colab
- Kaggle
Training in progress, step 50, checkpoint
Browse files
last-checkpoint/adapter_model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 69839888
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:b2f52210b2cf87e0ccd1dd365c144ff12b98d432d03d1182d05b04c027944b56
|
| 3 |
size 69839888
|
last-checkpoint/optimizer.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 139961199
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:dcc65ca6791256127638e354db5c80aaa7c3c229250f9fb2c6316327c20cad73
|
| 3 |
size 139961199
|
last-checkpoint/rng_state.pth
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 14645
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4373c49731cc6142e94702f0818c32829d15507599cd9c38571955e14b0c337e
|
| 3 |
size 14645
|
last-checkpoint/scheduler.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 1465
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:4d06a4f6b5bf353f960c09cd8473cd505cec37e78e2c01e0308679bf4aaaa00e
|
| 3 |
size 1465
|
last-checkpoint/trainer_state.json
CHANGED
|
@@ -1,10 +1,10 @@
|
|
| 1 |
{
|
| 2 |
-
"best_global_step":
|
| 3 |
-
"best_metric": 0.
|
| 4 |
-
"best_model_checkpoint": "./c2c_gemma4_e4b_qlora_checkpoint/checkpoint-
|
| 5 |
-
"epoch": 0.
|
| 6 |
"eval_steps": 25,
|
| 7 |
-
"global_step":
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
@@ -39,6 +39,37 @@
|
|
| 39 |
"eval_samples_per_second": 0.715,
|
| 40 |
"eval_steps_per_second": 0.715,
|
| 41 |
"step": 25
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 42 |
}
|
| 43 |
],
|
| 44 |
"logging_steps": 12,
|
|
@@ -58,7 +89,7 @@
|
|
| 58 |
"attributes": {}
|
| 59 |
}
|
| 60 |
},
|
| 61 |
-
"total_flos":
|
| 62 |
"train_batch_size": 1,
|
| 63 |
"trial_name": null,
|
| 64 |
"trial_params": null
|
|
|
|
| 1 |
{
|
| 2 |
+
"best_global_step": 50,
|
| 3 |
+
"best_metric": 0.00427626259624958,
|
| 4 |
+
"best_model_checkpoint": "./c2c_gemma4_e4b_qlora_checkpoint/checkpoint-50",
|
| 5 |
+
"epoch": 0.5,
|
| 6 |
"eval_steps": 25,
|
| 7 |
+
"global_step": 50,
|
| 8 |
"is_hyper_param_search": false,
|
| 9 |
"is_local_process_zero": true,
|
| 10 |
"is_world_process_zero": true,
|
|
|
|
| 39 |
"eval_samples_per_second": 0.715,
|
| 40 |
"eval_steps_per_second": 0.715,
|
| 41 |
"step": 25
|
| 42 |
+
},
|
| 43 |
+
{
|
| 44 |
+
"entropy": 1.0095942709594965,
|
| 45 |
+
"epoch": 0.36,
|
| 46 |
+
"grad_norm": 0.1298828125,
|
| 47 |
+
"learning_rate": 0.00018213058419243986,
|
| 48 |
+
"loss": 0.14251227180163065,
|
| 49 |
+
"mean_token_accuracy": 0.9945974703878164,
|
| 50 |
+
"num_tokens": 64401.0,
|
| 51 |
+
"step": 36
|
| 52 |
+
},
|
| 53 |
+
{
|
| 54 |
+
"entropy": 1.0027960228423278,
|
| 55 |
+
"epoch": 0.48,
|
| 56 |
+
"grad_norm": 0.09423828125,
|
| 57 |
+
"learning_rate": 0.0001738831615120275,
|
| 58 |
+
"loss": 0.06836261848608653,
|
| 59 |
+
"mean_token_accuracy": 0.9983609306315581,
|
| 60 |
+
"num_tokens": 86300.0,
|
| 61 |
+
"step": 48
|
| 62 |
+
},
|
| 63 |
+
{
|
| 64 |
+
"epoch": 0.5,
|
| 65 |
+
"eval_entropy": 0.9924230831861496,
|
| 66 |
+
"eval_loss": 0.00427626259624958,
|
| 67 |
+
"eval_mean_token_accuracy": 0.9989724820852279,
|
| 68 |
+
"eval_num_tokens": 89986.0,
|
| 69 |
+
"eval_runtime": 278.956,
|
| 70 |
+
"eval_samples_per_second": 0.717,
|
| 71 |
+
"eval_steps_per_second": 0.717,
|
| 72 |
+
"step": 50
|
| 73 |
}
|
| 74 |
],
|
| 75 |
"logging_steps": 12,
|
|
|
|
| 89 |
"attributes": {}
|
| 90 |
}
|
| 91 |
},
|
| 92 |
+
"total_flos": 2422236726599040.0,
|
| 93 |
"train_batch_size": 1,
|
| 94 |
"trial_name": null,
|
| 95 |
"trial_params": null
|