Mini.Deep.Thinker.11m | 20,000 new examples | step 1,250
Browse files- .gitattributes +1 -0
- README.md +3 -3
- model.safetensors +1 -1
- progress.json +22 -22
- seen_examples.jsonl +0 -0
- session_00001250.json +0 -0
- training_metadata.json +3 -3
- training_state.pt +2 -2
.gitattributes
CHANGED
|
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
|
|
|
|
|
| 33 |
*.zip filter=lfs diff=lfs merge=lfs -text
|
| 34 |
*.zst filter=lfs diff=lfs merge=lfs -text
|
| 35 |
*tfevents* filter=lfs diff=lfs merge=lfs -text
|
| 36 |
+
seen_examples.jsonl filter=lfs diff=lfs merge=lfs -text
|
README.md
CHANGED
|
@@ -63,9 +63,9 @@ Training state is persisted to Hugging Face, including:
|
|
| 63 |
|
| 64 |
## Current progress
|
| 65 |
|
| 66 |
-
- Unique examples trained:
|
| 67 |
-
- Global optimizer steps:
|
| 68 |
- Last session: 20,000 examples
|
| 69 |
-
- Best session loss:
|
| 70 |
|
| 71 |
This is an experimental small language model and is not guaranteed to produce factually or logically correct outputs.
|
|
|
|
| 63 |
|
| 64 |
## Current progress
|
| 65 |
|
| 66 |
+
- Unique examples trained: 40,000
|
| 67 |
+
- Global optimizer steps: 1,250
|
| 68 |
- Last session: 20,000 examples
|
| 69 |
+
- Best session loss: 1.88992
|
| 70 |
|
| 71 |
This is an experimental small language model and is not guaranteed to produce factually or logically correct outputs.
|
model.safetensors
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
size 49794720
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:1be18bc9af10fd3159e514d02469f6b5521f1176281be6130bf642829610e5eb
|
| 3 |
size 49794720
|
progress.json
CHANGED
|
@@ -5,36 +5,36 @@
|
|
| 5 |
"context_length": 1096,
|
| 6 |
"vocab_size": 4096,
|
| 7 |
"parameter_count": 11004896,
|
| 8 |
-
"global_step":
|
| 9 |
-
"unique_examples_completed":
|
| 10 |
"last_session_examples": 20000,
|
| 11 |
"max_session_examples": 20000,
|
| 12 |
"last_session_category_distribution": {
|
| 13 |
-
"
|
| 14 |
-
"
|
| 15 |
-
"
|
| 16 |
-
"
|
| 17 |
-
"Chat":
|
| 18 |
-
"
|
| 19 |
-
"Tool_Calling":
|
| 20 |
},
|
| 21 |
"last_session_source_distribution": {
|
| 22 |
-
"Plans11/Organized_PreTrain_WIUAI_1.3M":
|
| 23 |
-
"Plans11/Organized_PreTrain_Cyber_Security_640k":
|
| 24 |
-
"Plans11/Organized_PreTrain_Instruct_366k":
|
| 25 |
-
"Plans11/
|
| 26 |
-
"Plans11/
|
| 27 |
-
"Plans11/
|
| 28 |
-
"Plans11/
|
| 29 |
-
"Plans11/
|
| 30 |
-
"Plans11/
|
| 31 |
-
"Plans11/Organized_PreTrain_Agent_287k":
|
| 32 |
-
"Plans11/
|
| 33 |
},
|
| 34 |
-
"timestamp":
|
| 35 |
"causal_next_token": true,
|
| 36 |
"strict_example_context": true,
|
| 37 |
"tokenizer_immutable": true,
|
| 38 |
"shuffled_deterministic_scan": true,
|
| 39 |
-
"best_session_loss":
|
| 40 |
}
|
|
|
|
| 5 |
"context_length": 1096,
|
| 6 |
"vocab_size": 4096,
|
| 7 |
"parameter_count": 11004896,
|
| 8 |
+
"global_step": 1250,
|
| 9 |
+
"unique_examples_completed": 40000,
|
| 10 |
"last_session_examples": 20000,
|
| 11 |
"max_session_examples": 20000,
|
| 12 |
"last_session_category_distribution": {
|
| 13 |
+
"Code_Instruct": 12741,
|
| 14 |
+
"Instruct": 4926,
|
| 15 |
+
"Thought": 179,
|
| 16 |
+
"Think": 1607,
|
| 17 |
+
"Chat": 326,
|
| 18 |
+
"Reasoning": 176,
|
| 19 |
+
"Tool_Calling": 45
|
| 20 |
},
|
| 21 |
"last_session_source_distribution": {
|
| 22 |
+
"Plans11/Organized_PreTrain_WIUAI_1.3M": 10364,
|
| 23 |
+
"Plans11/Organized_PreTrain_Cyber_Security_640k": 4991,
|
| 24 |
+
"Plans11/Organized_PreTrain_Instruct_366k": 2789,
|
| 25 |
+
"Plans11/Organized_PreTrain_Frontier_Traces_150k": 523,
|
| 26 |
+
"Plans11/Organized_Pretrain_HR_213K": 243,
|
| 27 |
+
"Plans11/Organized_PreTrain_Frontier_Traces_47k": 341,
|
| 28 |
+
"Plans11/Organized_PreTrain_Persona_GOD_Seed_52K": 407,
|
| 29 |
+
"Plans11/Organized_PreTrain_Frontier_Traces_34K": 98,
|
| 30 |
+
"Plans11/Organized_PreTrain_Coding_239k": 137,
|
| 31 |
+
"Plans11/Organized_PreTrain_Agent_287k": 45,
|
| 32 |
+
"Plans11/Organized_PreTrain_Frontier_Traces_33k": 62
|
| 33 |
},
|
| 34 |
+
"timestamp": 1787150298.2830229,
|
| 35 |
"causal_next_token": true,
|
| 36 |
"strict_example_context": true,
|
| 37 |
"tokenizer_immutable": true,
|
| 38 |
"shuffled_deterministic_scan": true,
|
| 39 |
+
"best_session_loss": 1.889919606000185
|
| 40 |
}
|
seen_examples.jsonl
CHANGED
|
The diff for this file is too large to render.
See raw diff
|
|
|
session_00001250.json
ADDED
|
The diff for this file is too large to render.
See raw diff
|
|
|
training_metadata.json
CHANGED
|
@@ -6,8 +6,8 @@
|
|
| 6 |
"strict_example_limit": 1095,
|
| 7 |
"vocab_size": 4096,
|
| 8 |
"dataset": "Plans11/Organized_PreTrain_1k_Context",
|
| 9 |
-
"global_step":
|
| 10 |
-
"unique_examples_trained":
|
| 11 |
"last_session_size": 20000,
|
| 12 |
"learning_rate": 0.0003,
|
| 13 |
"batch_size": 8,
|
|
@@ -24,5 +24,5 @@
|
|
| 24 |
"tokenizer_immutable": true,
|
| 25 |
"content_hash_deduplication": true,
|
| 26 |
"shuffled_deterministic_scan": true,
|
| 27 |
-
"best_session_loss":
|
| 28 |
}
|
|
|
|
| 6 |
"strict_example_limit": 1095,
|
| 7 |
"vocab_size": 4096,
|
| 8 |
"dataset": "Plans11/Organized_PreTrain_1k_Context",
|
| 9 |
+
"global_step": 1250,
|
| 10 |
+
"unique_examples_trained": 40000,
|
| 11 |
"last_session_size": 20000,
|
| 12 |
"learning_rate": 0.0003,
|
| 13 |
"batch_size": 8,
|
|
|
|
| 24 |
"tokenizer_immutable": true,
|
| 25 |
"content_hash_deduplication": true,
|
| 26 |
"shuffled_deterministic_scan": true,
|
| 27 |
+
"best_session_loss": 1.889919606000185
|
| 28 |
}
|
training_state.pt
CHANGED
|
@@ -1,3 +1,3 @@
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
-
oid sha256:
|
| 3 |
-
size
|
|
|
|
| 1 |
version https://git-lfs.github.com/spec/v1
|
| 2 |
+
oid sha256:066eea6aa8152a4365b78c54d9e3f2ddc025f7acdea6b2b93a19b6da17a1eb66
|
| 3 |
+
size 132169595
|