Plans11 commited on
Commit
b1c0908
·
verified ·
1 Parent(s): c33e332

Mini.Deep.Thinker.11m | 20,000 new examples | step 1,250

Browse files
.gitattributes CHANGED
@@ -33,3 +33,4 @@ saved_model/**/* filter=lfs diff=lfs merge=lfs -text
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
 
 
33
  *.zip filter=lfs diff=lfs merge=lfs -text
34
  *.zst filter=lfs diff=lfs merge=lfs -text
35
  *tfevents* filter=lfs diff=lfs merge=lfs -text
36
+ seen_examples.jsonl filter=lfs diff=lfs merge=lfs -text
README.md CHANGED
@@ -63,9 +63,9 @@ Training state is persisted to Hugging Face, including:
63
 
64
  ## Current progress
65
 
66
- - Unique examples trained: 20,000
67
- - Global optimizer steps: 625
68
  - Last session: 20,000 examples
69
- - Best session loss: 3.06134
70
 
71
  This is an experimental small language model and is not guaranteed to produce factually or logically correct outputs.
 
63
 
64
  ## Current progress
65
 
66
+ - Unique examples trained: 40,000
67
+ - Global optimizer steps: 1,250
68
  - Last session: 20,000 examples
69
+ - Best session loss: 1.88992
70
 
71
  This is an experimental small language model and is not guaranteed to produce factually or logically correct outputs.
model.safetensors CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:7bc26d2dd56ce812cd2ea3f039fa6bd5e2e476d09779360113b7dc58ac353267
3
  size 49794720
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:1be18bc9af10fd3159e514d02469f6b5521f1176281be6130bf642829610e5eb
3
  size 49794720
progress.json CHANGED
@@ -5,36 +5,36 @@
5
  "context_length": 1096,
6
  "vocab_size": 4096,
7
  "parameter_count": 11004896,
8
- "global_step": 625,
9
- "unique_examples_completed": 20000,
10
  "last_session_examples": 20000,
11
  "max_session_examples": 20000,
12
  "last_session_category_distribution": {
13
- "Instruct": 4952,
14
- "Code_Instruct": 12819,
15
- "Think": 1486,
16
- "Reasoning": 165,
17
- "Chat": 332,
18
- "Thought": 205,
19
- "Tool_Calling": 41
20
  },
21
  "last_session_source_distribution": {
22
- "Plans11/Organized_PreTrain_WIUAI_1.3M": 10290,
23
- "Plans11/Organized_PreTrain_Cyber_Security_640k": 5077,
24
- "Plans11/Organized_PreTrain_Instruct_366k": 2809,
25
- "Plans11/Organized_PreTrain_Frontier_Traces_47k": 311,
26
- "Plans11/Organized_PreTrain_Frontier_Traces_150k": 494,
27
- "Plans11/Organized_Pretrain_HR_213K": 254,
28
- "Plans11/Organized_PreTrain_Frontier_Traces_34K": 119,
29
- "Plans11/Organized_PreTrain_Persona_GOD_Seed_52K": 413,
30
- "Plans11/Organized_PreTrain_Frontier_Traces_33k": 52,
31
- "Plans11/Organized_PreTrain_Agent_287k": 41,
32
- "Plans11/Organized_PreTrain_Coding_239k": 140
33
  },
34
- "timestamp": 1787146729.219449,
35
  "causal_next_token": true,
36
  "strict_example_context": true,
37
  "tokenizer_immutable": true,
38
  "shuffled_deterministic_scan": true,
39
- "best_session_loss": 3.061336768269539
40
  }
 
5
  "context_length": 1096,
6
  "vocab_size": 4096,
7
  "parameter_count": 11004896,
8
+ "global_step": 1250,
9
+ "unique_examples_completed": 40000,
10
  "last_session_examples": 20000,
11
  "max_session_examples": 20000,
12
  "last_session_category_distribution": {
13
+ "Code_Instruct": 12741,
14
+ "Instruct": 4926,
15
+ "Thought": 179,
16
+ "Think": 1607,
17
+ "Chat": 326,
18
+ "Reasoning": 176,
19
+ "Tool_Calling": 45
20
  },
21
  "last_session_source_distribution": {
22
+ "Plans11/Organized_PreTrain_WIUAI_1.3M": 10364,
23
+ "Plans11/Organized_PreTrain_Cyber_Security_640k": 4991,
24
+ "Plans11/Organized_PreTrain_Instruct_366k": 2789,
25
+ "Plans11/Organized_PreTrain_Frontier_Traces_150k": 523,
26
+ "Plans11/Organized_Pretrain_HR_213K": 243,
27
+ "Plans11/Organized_PreTrain_Frontier_Traces_47k": 341,
28
+ "Plans11/Organized_PreTrain_Persona_GOD_Seed_52K": 407,
29
+ "Plans11/Organized_PreTrain_Frontier_Traces_34K": 98,
30
+ "Plans11/Organized_PreTrain_Coding_239k": 137,
31
+ "Plans11/Organized_PreTrain_Agent_287k": 45,
32
+ "Plans11/Organized_PreTrain_Frontier_Traces_33k": 62
33
  },
34
+ "timestamp": 1787150298.2830229,
35
  "causal_next_token": true,
36
  "strict_example_context": true,
37
  "tokenizer_immutable": true,
38
  "shuffled_deterministic_scan": true,
39
+ "best_session_loss": 1.889919606000185
40
  }
seen_examples.jsonl CHANGED
The diff for this file is too large to render. See raw diff
 
session_00001250.json ADDED
The diff for this file is too large to render. See raw diff
 
training_metadata.json CHANGED
@@ -6,8 +6,8 @@
6
  "strict_example_limit": 1095,
7
  "vocab_size": 4096,
8
  "dataset": "Plans11/Organized_PreTrain_1k_Context",
9
- "global_step": 625,
10
- "unique_examples_trained": 20000,
11
  "last_session_size": 20000,
12
  "learning_rate": 0.0003,
13
  "batch_size": 8,
@@ -24,5 +24,5 @@
24
  "tokenizer_immutable": true,
25
  "content_hash_deduplication": true,
26
  "shuffled_deterministic_scan": true,
27
- "best_session_loss": 3.061336768269539
28
  }
 
6
  "strict_example_limit": 1095,
7
  "vocab_size": 4096,
8
  "dataset": "Plans11/Organized_PreTrain_1k_Context",
9
+ "global_step": 1250,
10
+ "unique_examples_trained": 40000,
11
  "last_session_size": 20000,
12
  "learning_rate": 0.0003,
13
  "batch_size": 8,
 
24
  "tokenizer_immutable": true,
25
  "content_hash_deduplication": true,
26
  "shuffled_deterministic_scan": true,
27
+ "best_session_loss": 1.889919606000185
28
  }
training_state.pt CHANGED
@@ -1,3 +1,3 @@
1
  version https://git-lfs.github.com/spec/v1
2
- oid sha256:0fc392f4b98a53da2d7d1d54e0406f7b303e0f825e90604150d6de6d104a553f
3
- size 132168635
 
1
  version https://git-lfs.github.com/spec/v1
2
+ oid sha256:066eea6aa8152a4365b78c54d9e3f2ddc025f7acdea6b2b93a19b6da17a1eb66
3
+ size 132169595