Upload sft_01_collect_hf_datasets.py with huggingface_hub
Browse files- sft_01_collect_hf_datasets.py +1302 -0
sft_01_collect_hf_datasets.py
ADDED
|
@@ -0,0 +1,1302 @@
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
|
| 1 |
+
"""
|
| 2 |
+
V5-SFT Asama 1: HuggingFace Turkce instruction datasetlerini topla + filtrele.
|
| 3 |
+
|
| 4 |
+
150K hedef, 25+ dataset, 6 kategori:
|
| 5 |
+
- General Instruction (40K)
|
| 6 |
+
- Reasoning (45K)
|
| 7 |
+
- Knowledge/QA (30K)
|
| 8 |
+
- Conversation & Cultural (20K)
|
| 9 |
+
- Quality Curated (10K)
|
| 10 |
+
- Function Calling (5K)
|
| 11 |
+
|
| 12 |
+
Cikti:
|
| 13 |
+
data/sft/01_collected.jsonl (filter sonrasi, ChatML format)
|
| 14 |
+
data/sft/01_stats.json (her dataset icin istatistik)
|
| 15 |
+
data/sft/01_rejected_samples.jsonl (incelemek icin)
|
| 16 |
+
|
| 17 |
+
Kullanim:
|
| 18 |
+
python sft_01_collect_hf_datasets.py # 150K hedef
|
| 19 |
+
python sft_01_collect_hf_datasets.py --target 50000 # daha az
|
| 20 |
+
python sft_01_collect_hf_datasets.py --categories reasoning # tek kategori
|
| 21 |
+
python sft_01_collect_hf_datasets.py --datasets turkish_mmlu # tek dataset
|
| 22 |
+
python sft_01_collect_hf_datasets.py --skip-failed # erisilemeyenleri atla
|
| 23 |
+
"""
|
| 24 |
+
|
| 25 |
+
import argparse
|
| 26 |
+
import hashlib
|
| 27 |
+
import json
|
| 28 |
+
import os
|
| 29 |
+
import random
|
| 30 |
+
import re
|
| 31 |
+
import sys
|
| 32 |
+
import warnings
|
| 33 |
+
from collections import Counter, defaultdict
|
| 34 |
+
from pathlib import Path
|
| 35 |
+
|
| 36 |
+
os.environ["HF_HUB_DISABLE_SYMLINKS_WARNING"] = "1"
|
| 37 |
+
warnings.filterwarnings("ignore", category=UserWarning, module="huggingface_hub")
|
| 38 |
+
|
| 39 |
+
try:
|
| 40 |
+
from datasets import load_dataset, get_dataset_config_names
|
| 41 |
+
from tqdm import tqdm
|
| 42 |
+
except ImportError:
|
| 43 |
+
print("! Bagimliliklar eksik. Yukle:")
|
| 44 |
+
print(" pip install datasets tqdm")
|
| 45 |
+
sys.exit(1)
|
| 46 |
+
|
| 47 |
+
DATA_DIR = Path(__file__).parent / "data" / "sft"
|
| 48 |
+
CACHE_DIR = DATA_DIR / "cache"
|
| 49 |
+
DATA_DIR.mkdir(parents=True, exist_ok=True)
|
| 50 |
+
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
| 51 |
+
|
| 52 |
+
# =====================================================================
|
| 53 |
+
# Dataset Konfigurasyonlari (Kategori bazli)
|
| 54 |
+
# =====================================================================
|
| 55 |
+
DATASETS = {
|
| 56 |
+
# ========== GENERAL INSTRUCTION (40K) ==========
|
| 57 |
+
# instructurca DROPPED — kalitesiz (broken Turkce-translated code, weird prompts).
|
| 58 |
+
# Yerine 4 yuksek kaliteli kaynak (toplam 25K):
|
| 59 |
+
"fineweb_augmented_tr": {
|
| 60 |
+
"repo": "Ba2han/UltraFineWeb-Augmented_TR_Instruct",
|
| 61 |
+
"split": "train",
|
| 62 |
+
"category": "general",
|
| 63 |
+
"target": 14000, # eksik 4K'yi buradan kapat (kaliteli kaynak)
|
| 64 |
+
"is_messages": True,
|
| 65 |
+
},
|
| 66 |
+
"quardo_gpt4o_tr": {
|
| 67 |
+
"repo": "Quardo/Turkish-Chat_GPT-4O",
|
| 68 |
+
"split": "train",
|
| 69 |
+
"category": "general",
|
| 70 |
+
"target": 6000,
|
| 71 |
+
"is_quardo": True,
|
| 72 |
+
},
|
| 73 |
+
"alpaca_evol_tr": {
|
| 74 |
+
"repo": "malhajar/alpaca-evol-instruct-turkish",
|
| 75 |
+
"split": "train",
|
| 76 |
+
"category": "general",
|
| 77 |
+
"target": 5000,
|
| 78 |
+
"input_keys": ["evolved_instruct", "instruction"],
|
| 79 |
+
"output_keys": ["response", "output"],
|
| 80 |
+
},
|
| 81 |
+
"gpteacher_tr": {
|
| 82 |
+
"repo": "umarigan/GPTeacher-General-Instruct-tr",
|
| 83 |
+
"split": "train",
|
| 84 |
+
"category": "general",
|
| 85 |
+
"target": 3000,
|
| 86 |
+
"input_keys": ["instruction"],
|
| 87 |
+
"input2_keys": ["input"],
|
| 88 |
+
"output_keys": ["response", "output"],
|
| 89 |
+
# Ceviri gorevlerinde geri-ceviri hatasi var, asst min uzunluk yukseltildi
|
| 90 |
+
"min_asst_len": 100,
|
| 91 |
+
"skip_translation": True,
|
| 92 |
+
},
|
| 93 |
+
"alpaca_gpt4": {
|
| 94 |
+
"repo": "malhajar/alpaca-gpt4-tr",
|
| 95 |
+
"split": "train",
|
| 96 |
+
"category": "general",
|
| 97 |
+
"target": 10000,
|
| 98 |
+
# Turkish fields have -turkish suffix
|
| 99 |
+
"input_keys": ["instruction-turkish"],
|
| 100 |
+
"input2_keys": ["input-turkish"],
|
| 101 |
+
"output_keys": ["output-turkish"],
|
| 102 |
+
},
|
| 103 |
+
"dolly_15k_tr": {
|
| 104 |
+
"repo": "atasoglu/databricks-dolly-15k-tr",
|
| 105 |
+
"split": "train",
|
| 106 |
+
"category": "general",
|
| 107 |
+
"target": 5000,
|
| 108 |
+
"input_keys": ["instruction", "soru"],
|
| 109 |
+
"input2_keys": ["context", "input"],
|
| 110 |
+
"output_keys": ["response", "output", "cevap"],
|
| 111 |
+
},
|
| 112 |
+
"no_robots_tr": {
|
| 113 |
+
"repo": "beratcmn/no_robots_turkish",
|
| 114 |
+
"split": "train",
|
| 115 |
+
"category": "general",
|
| 116 |
+
"target": 5000,
|
| 117 |
+
# Translation fields
|
| 118 |
+
"input_keys": ["question_translation"],
|
| 119 |
+
"output_keys": ["answer_translation"],
|
| 120 |
+
},
|
| 121 |
+
|
| 122 |
+
# ========== REASONING (45K) ⭐ ==========
|
| 123 |
+
"thinking_data_200k": {
|
| 124 |
+
"repo": "erythropygia/ThinkingData-200K-Turkish",
|
| 125 |
+
"split": "train",
|
| 126 |
+
"category": "reasoning",
|
| 127 |
+
"target": 18000,
|
| 128 |
+
# Schema: messages list (system+user) + reasoning + answer
|
| 129 |
+
"is_thinking_data": True,
|
| 130 |
+
},
|
| 131 |
+
"metamathqa_tr": {
|
| 132 |
+
"repo": "onur48/MetaMathQA-Turkish-corrected",
|
| 133 |
+
"split": "train",
|
| 134 |
+
"category": "reasoning",
|
| 135 |
+
"target": 12000,
|
| 136 |
+
"input_keys": ["query", "question", "problem", "input"],
|
| 137 |
+
"output_keys": ["response", "output", "answer"],
|
| 138 |
+
},
|
| 139 |
+
"gsm8k_tr": {
|
| 140 |
+
"repo": "ytu-ce-cosmos/gsm8k_tr",
|
| 141 |
+
"split": "train",
|
| 142 |
+
"category": "reasoning",
|
| 143 |
+
"target": 8000,
|
| 144 |
+
"input_keys": ["question", "input", "soru"],
|
| 145 |
+
"output_keys": ["answer", "output", "cevap"],
|
| 146 |
+
},
|
| 147 |
+
"openthoughts_tr": {
|
| 148 |
+
"repo": "selimc/OpenThoughts-TR-18k",
|
| 149 |
+
"split": "train",
|
| 150 |
+
"category": "reasoning",
|
| 151 |
+
"target": 4000,
|
| 152 |
+
"input_keys": ["prompt"],
|
| 153 |
+
"output_keys": ["response"],
|
| 154 |
+
},
|
| 155 |
+
"turkish_medical_reasoning": {
|
| 156 |
+
"repo": "ituperceptron/turkish_medical_reasoning",
|
| 157 |
+
"split": "train",
|
| 158 |
+
"category": "reasoning",
|
| 159 |
+
"target": 2000,
|
| 160 |
+
"input_keys": ["question"],
|
| 161 |
+
"output_keys": ["answer_content"],
|
| 162 |
+
"thinking_key": "reasoning_content", # ayri thinking alani
|
| 163 |
+
},
|
| 164 |
+
"tree_of_thought_tr": {
|
| 165 |
+
"repo": "emre/ct_tree_of_thought_turkish",
|
| 166 |
+
"split": "train",
|
| 167 |
+
"category": "reasoning",
|
| 168 |
+
"target": 600,
|
| 169 |
+
# Special handler — chat_format alani var (JSON)
|
| 170 |
+
"is_tot": True,
|
| 171 |
+
},
|
| 172 |
+
"medium_math_tr": {
|
| 173 |
+
"repo": "erayalp/medium_turkish_math_reasoning",
|
| 174 |
+
"split": "train",
|
| 175 |
+
"category": "reasoning",
|
| 176 |
+
"target": 1000,
|
| 177 |
+
"input_keys": ["question", "problem", "instruction"],
|
| 178 |
+
"output_keys": ["answer", "solution", "response"],
|
| 179 |
+
},
|
| 180 |
+
|
| 181 |
+
# ========== KNOWLEDGE / QA (30K) ⭐ ==========
|
| 182 |
+
# turkish_mmlu gated — skip (kullanici elle erisim isteyebilir)
|
| 183 |
+
"turkish_exam": {
|
| 184 |
+
"repo": "bezir/turkish_exam_instructions",
|
| 185 |
+
"split": "train",
|
| 186 |
+
"category": "knowledge",
|
| 187 |
+
"target": 6000,
|
| 188 |
+
"input_keys": ["instruction", "soru", "question", "prompt"],
|
| 189 |
+
"output_keys": ["output", "response", "cevap", "answer"],
|
| 190 |
+
},
|
| 191 |
+
"wikirag_tr": {
|
| 192 |
+
"repo": "Metin/WikiRAG-TR",
|
| 193 |
+
"split": "train",
|
| 194 |
+
"category": "knowledge",
|
| 195 |
+
"target": 5000,
|
| 196 |
+
"input_keys": ["soru", "question", "instruction"],
|
| 197 |
+
"input2_keys": ["context", "passage"],
|
| 198 |
+
"output_keys": ["cevap", "answer", "response", "output"],
|
| 199 |
+
},
|
| 200 |
+
"instruct_papers_tr": {
|
| 201 |
+
"repo": "selimc/InstructPapers-TR",
|
| 202 |
+
"split": "train",
|
| 203 |
+
"category": "knowledge",
|
| 204 |
+
"target": 2000,
|
| 205 |
+
"input_keys": ["instruction", "question", "soru"],
|
| 206 |
+
"input2_keys": ["context", "abstract"],
|
| 207 |
+
"output_keys": ["output", "response", "answer"],
|
| 208 |
+
},
|
| 209 |
+
"truthful_qa_tr": {
|
| 210 |
+
"repo": "malhajar/truthfull_qa-tr",
|
| 211 |
+
"config_name": "generation", # config secimi gerekli
|
| 212 |
+
"split": "validation", # truthful_qa'da train yok, validation var
|
| 213 |
+
"category": "knowledge",
|
| 214 |
+
"target": 800,
|
| 215 |
+
"input_keys": ["question", "soru"],
|
| 216 |
+
"output_keys": ["best_answer", "correct_answers", "answer"],
|
| 217 |
+
},
|
| 218 |
+
"gpqa_tr": {
|
| 219 |
+
"repo": "ytu-ce-cosmos/gpqa-extended_tr",
|
| 220 |
+
"split": "train",
|
| 221 |
+
"category": "knowledge",
|
| 222 |
+
"target": 500,
|
| 223 |
+
# Special schema: Question + Correct Answer + Incorrect Answers 1-3
|
| 224 |
+
"is_gpqa": True,
|
| 225 |
+
},
|
| 226 |
+
"finance_qa": {
|
| 227 |
+
"repo": "umarigan/turkiye_finance_qa",
|
| 228 |
+
"split": "train",
|
| 229 |
+
"category": "knowledge",
|
| 230 |
+
"target": 400,
|
| 231 |
+
"input_keys": ["question", "soru", "instruction"],
|
| 232 |
+
"output_keys": ["answer", "cevap", "response", "output"],
|
| 233 |
+
},
|
| 234 |
+
|
| 235 |
+
# ========== CONVERSATION & CULTURAL (20K) ⭐ ==========
|
| 236 |
+
# oasst1_tr — tree format kompleks, skip (TR sample az)
|
| 237 |
+
"turkish_poems": {
|
| 238 |
+
"repo": "beratcmn/instruction-turkish-poems",
|
| 239 |
+
"split": "train",
|
| 240 |
+
"category": "conversation",
|
| 241 |
+
"target": 4000,
|
| 242 |
+
"input_keys": ["instruction"],
|
| 243 |
+
"output_keys": ["poem"],
|
| 244 |
+
},
|
| 245 |
+
"turkish_recipes": {
|
| 246 |
+
"repo": "mertbozkurt/llama2-TR-recipe",
|
| 247 |
+
"split": "train",
|
| 248 |
+
"category": "conversation",
|
| 249 |
+
"target": 4000,
|
| 250 |
+
# Special: text field has [INST]...[/INST] format
|
| 251 |
+
"is_inst_format": True,
|
| 252 |
+
},
|
| 253 |
+
"everyday_conv_tr": {
|
| 254 |
+
"repo": "SoAp9035/everyday-conversations-tur",
|
| 255 |
+
"split": "train",
|
| 256 |
+
"category": "conversation",
|
| 257 |
+
"target": 2000,
|
| 258 |
+
"input_keys": ["messages", "instruction", "prompt"],
|
| 259 |
+
"output_keys": ["response", "output"],
|
| 260 |
+
"is_messages": True,
|
| 261 |
+
},
|
| 262 |
+
"soap_instructions": {
|
| 263 |
+
"repo": "SoAp9035/turkish_instructions",
|
| 264 |
+
"split": "train",
|
| 265 |
+
"category": "conversation",
|
| 266 |
+
"target": 1000,
|
| 267 |
+
"input_keys": ["user"],
|
| 268 |
+
"output_keys": ["assistant"],
|
| 269 |
+
},
|
| 270 |
+
"masallar": {
|
| 271 |
+
"repo": "umutphp/masallar",
|
| 272 |
+
"split": "train",
|
| 273 |
+
"category": "conversation",
|
| 274 |
+
"target": 1000,
|
| 275 |
+
"text_keys": ["text", "story", "content"],
|
| 276 |
+
"is_essay": True,
|
| 277 |
+
},
|
| 278 |
+
|
| 279 |
+
# ========== QUALITY CURATED (10K) ==========
|
| 280 |
+
"aya_tr": {
|
| 281 |
+
"repo": "sayhan/aya_dataset_tur",
|
| 282 |
+
"split": "train",
|
| 283 |
+
"category": "curated",
|
| 284 |
+
"target": 4000,
|
| 285 |
+
"input_keys": ["inputs", "instruction", "prompt"],
|
| 286 |
+
"output_keys": ["targets", "output", "response"],
|
| 287 |
+
},
|
| 288 |
+
"alican_sft": {
|
| 289 |
+
"repo": "AlicanKiraz0/Turkish-SFT-Dataset-v1.0",
|
| 290 |
+
"split": "train",
|
| 291 |
+
"category": "curated",
|
| 292 |
+
"target": 4000,
|
| 293 |
+
# Direct ChatML: system/user/assistant
|
| 294 |
+
"input_keys": ["user"],
|
| 295 |
+
"output_keys": ["assistant"],
|
| 296 |
+
"system_keys": ["system"],
|
| 297 |
+
},
|
| 298 |
+
"lima_tr": {
|
| 299 |
+
"repo": "beratcmn/lima-tr",
|
| 300 |
+
"split": "train",
|
| 301 |
+
"category": "curated",
|
| 302 |
+
"target": 1500,
|
| 303 |
+
# Translated fields
|
| 304 |
+
"input_keys": ["translated"],
|
| 305 |
+
"output_keys": ["translated_reply"],
|
| 306 |
+
},
|
| 307 |
+
"general_knowledge_qa": {
|
| 308 |
+
"repo": "nisancoskun/turkish_general_knowledge_qa",
|
| 309 |
+
"split": "train",
|
| 310 |
+
"category": "curated",
|
| 311 |
+
"target": 60,
|
| 312 |
+
"input_keys": ["question", "soru", "instruction"],
|
| 313 |
+
"output_keys": ["answer", "cevap", "response"],
|
| 314 |
+
},
|
| 315 |
+
|
| 316 |
+
# ========== FUNCTION CALLING (5K) ==========
|
| 317 |
+
"function_calling": {
|
| 318 |
+
"repo": "atasoglu/turkish-function-calling-20k",
|
| 319 |
+
"split": "train",
|
| 320 |
+
"category": "function",
|
| 321 |
+
"target": 3000,
|
| 322 |
+
"is_function_calling": True,
|
| 323 |
+
"tools_key": "tools",
|
| 324 |
+
"query_key": "query",
|
| 325 |
+
"answers_key": "answers",
|
| 326 |
+
"tag": "function_calling",
|
| 327 |
+
},
|
| 328 |
+
"tool_calling": {
|
| 329 |
+
"repo": "atasoglu/turkish-tool-calling-10k",
|
| 330 |
+
"split": "train",
|
| 331 |
+
"category": "function",
|
| 332 |
+
"target": 2000,
|
| 333 |
+
"is_function_calling": True,
|
| 334 |
+
"is_tool_calling": True,
|
| 335 |
+
"tag": "tool_calling",
|
| 336 |
+
},
|
| 337 |
+
}
|
| 338 |
+
|
| 339 |
+
CATEGORIES = ["general", "reasoning", "knowledge", "conversation", "curated", "function"]
|
| 340 |
+
|
| 341 |
+
|
| 342 |
+
# =====================================================================
|
| 343 |
+
# Helper Fonksiyonlar
|
| 344 |
+
# =====================================================================
|
| 345 |
+
def find_field(record: dict, candidates: list) -> str:
|
| 346 |
+
"""Recordda alan ara — basta/sonda bosluga tolerans."""
|
| 347 |
+
stripped_keys = {k.strip(): k for k in record.keys()}
|
| 348 |
+
for c in candidates:
|
| 349 |
+
c_strip = c.strip()
|
| 350 |
+
if c_strip in stripped_keys:
|
| 351 |
+
val = record[stripped_keys[c_strip]]
|
| 352 |
+
if isinstance(val, str) and val.strip():
|
| 353 |
+
return val.strip()
|
| 354 |
+
if isinstance(val, list) and val:
|
| 355 |
+
# Listede ilk string item al
|
| 356 |
+
for item in val:
|
| 357 |
+
if isinstance(item, str) and item.strip():
|
| 358 |
+
return item.strip()
|
| 359 |
+
return ""
|
| 360 |
+
|
| 361 |
+
|
| 362 |
+
def find_field_any(record: dict, candidates: list):
|
| 363 |
+
"""find_field ama list/dict de dondurur."""
|
| 364 |
+
stripped_keys = {k.strip(): k for k in record.keys()}
|
| 365 |
+
for c in candidates:
|
| 366 |
+
c_strip = c.strip()
|
| 367 |
+
if c_strip in stripped_keys:
|
| 368 |
+
val = record[stripped_keys[c_strip]]
|
| 369 |
+
if val:
|
| 370 |
+
return val
|
| 371 |
+
return None
|
| 372 |
+
|
| 373 |
+
|
| 374 |
+
# =====================================================================
|
| 375 |
+
# Extract — Source-specific handlers
|
| 376 |
+
# =====================================================================
|
| 377 |
+
def extract_messages_format(rec: dict, config: dict) -> dict | None:
|
| 378 |
+
"""messages: [{role, content}, ...] format."""
|
| 379 |
+
msgs = find_field_any(rec, ["messages", "conversations"])
|
| 380 |
+
if not isinstance(msgs, list) or len(msgs) < 2:
|
| 381 |
+
return None
|
| 382 |
+
user_msg = None
|
| 383 |
+
asst_msg = None
|
| 384 |
+
sys_msg = None
|
| 385 |
+
for m in msgs:
|
| 386 |
+
if not isinstance(m, dict):
|
| 387 |
+
continue
|
| 388 |
+
role = m.get("role") or m.get("from", "")
|
| 389 |
+
content = m.get("content") or m.get("value", "")
|
| 390 |
+
if not content:
|
| 391 |
+
continue
|
| 392 |
+
if role == "system":
|
| 393 |
+
sys_msg = content
|
| 394 |
+
elif role in ("user", "human"):
|
| 395 |
+
user_msg = content
|
| 396 |
+
elif role in ("assistant", "gpt", "bot"):
|
| 397 |
+
asst_msg = content
|
| 398 |
+
if not user_msg or not asst_msg:
|
| 399 |
+
return None
|
| 400 |
+
out = {"user": user_msg, "assistant": asst_msg}
|
| 401 |
+
if sys_msg:
|
| 402 |
+
out["system"] = sys_msg
|
| 403 |
+
return out
|
| 404 |
+
|
| 405 |
+
|
| 406 |
+
def extract_mmlu(rec: dict) -> dict | None:
|
| 407 |
+
"""MMLU MCQA: question + choices + answer."""
|
| 408 |
+
question = (rec.get("question") or rec.get("soru") or
|
| 409 |
+
rec.get("Soru") or rec.get("Question") or "")
|
| 410 |
+
if not question or len(question) < 10:
|
| 411 |
+
return None
|
| 412 |
+
|
| 413 |
+
# Choices farkli isimler
|
| 414 |
+
choices = None
|
| 415 |
+
for key in ["choices", "options", "secenekler", "Choices"]:
|
| 416 |
+
if key in rec:
|
| 417 |
+
choices = rec[key]
|
| 418 |
+
break
|
| 419 |
+
if choices is None:
|
| 420 |
+
# A/B/C/D ayri alanlar
|
| 421 |
+
opts = []
|
| 422 |
+
for letter in ["A", "B", "C", "D", "E"]:
|
| 423 |
+
for k in [letter, f"choice_{letter}", f"option_{letter}",
|
| 424 |
+
f"secenek_{letter}", letter.lower()]:
|
| 425 |
+
if k in rec and rec[k]:
|
| 426 |
+
opts.append(str(rec[k]))
|
| 427 |
+
break
|
| 428 |
+
if opts:
|
| 429 |
+
choices = opts
|
| 430 |
+
|
| 431 |
+
if not choices:
|
| 432 |
+
return None
|
| 433 |
+
if isinstance(choices, str):
|
| 434 |
+
# Bazi datasetlerde virgulle ayrilmis
|
| 435 |
+
choices = [c.strip() for c in choices.split("|") if c.strip()]
|
| 436 |
+
|
| 437 |
+
# Cevap — letter or index
|
| 438 |
+
answer = (rec.get("answer") or rec.get("correct") or
|
| 439 |
+
rec.get("cevap") or rec.get("Answer") or "")
|
| 440 |
+
|
| 441 |
+
if isinstance(answer, int):
|
| 442 |
+
if 0 <= answer < len(choices):
|
| 443 |
+
answer_text = choices[answer]
|
| 444 |
+
answer_letter = chr(65 + answer)
|
| 445 |
+
else:
|
| 446 |
+
return None
|
| 447 |
+
elif isinstance(answer, str):
|
| 448 |
+
answer = answer.strip()
|
| 449 |
+
if len(answer) == 1 and answer.upper() in "ABCDE":
|
| 450 |
+
idx = ord(answer.upper()) - 65
|
| 451 |
+
if idx < len(choices):
|
| 452 |
+
answer_text = choices[idx]
|
| 453 |
+
answer_letter = answer.upper()
|
| 454 |
+
else:
|
| 455 |
+
return None
|
| 456 |
+
else:
|
| 457 |
+
# Cevap dogrudan metin
|
| 458 |
+
answer_text = answer
|
| 459 |
+
answer_letter = "?"
|
| 460 |
+
else:
|
| 461 |
+
return None
|
| 462 |
+
|
| 463 |
+
# Format
|
| 464 |
+
choices_text = "\n".join(
|
| 465 |
+
f"{chr(65+i)}) {c}" for i, c in enumerate(choices)
|
| 466 |
+
)
|
| 467 |
+
user = f"{question}\n\n{choices_text}"
|
| 468 |
+
assistant = f"Cevap: {answer_letter}) {answer_text}"
|
| 469 |
+
return {"user": user, "assistant": assistant}
|
| 470 |
+
|
| 471 |
+
|
| 472 |
+
def _parse_maybe_json(val):
|
| 473 |
+
"""List/dict/string — hepsini handle et."""
|
| 474 |
+
if val is None:
|
| 475 |
+
return None
|
| 476 |
+
if isinstance(val, (list, dict)):
|
| 477 |
+
return val
|
| 478 |
+
if isinstance(val, str):
|
| 479 |
+
s = val.strip()
|
| 480 |
+
if s.startswith(("[", "{")):
|
| 481 |
+
try:
|
| 482 |
+
return json.loads(s)
|
| 483 |
+
except Exception:
|
| 484 |
+
return s
|
| 485 |
+
return s
|
| 486 |
+
return val
|
| 487 |
+
|
| 488 |
+
|
| 489 |
+
def _to_json_str(val):
|
| 490 |
+
"""Tum tipleri JSON string'e cevir."""
|
| 491 |
+
if val is None:
|
| 492 |
+
return ""
|
| 493 |
+
if isinstance(val, str):
|
| 494 |
+
return val
|
| 495 |
+
try:
|
| 496 |
+
return json.dumps(val, ensure_ascii=False)
|
| 497 |
+
except Exception:
|
| 498 |
+
return str(val)
|
| 499 |
+
|
| 500 |
+
|
| 501 |
+
def extract_function_calling(rec: dict, config: dict) -> dict | None:
|
| 502 |
+
"""tools + query + answers format."""
|
| 503 |
+
if config.get("is_tool_calling"):
|
| 504 |
+
# tool-calling-10k: messages list, assistant_calls list, tools list
|
| 505 |
+
msgs = _parse_maybe_json(rec.get("messages"))
|
| 506 |
+
calls = _parse_maybe_json(rec.get("assistant_calls"))
|
| 507 |
+
tools = _parse_maybe_json(rec.get("tools"))
|
| 508 |
+
|
| 509 |
+
# User mesajini bul
|
| 510 |
+
user_msg = None
|
| 511 |
+
if isinstance(msgs, list):
|
| 512 |
+
for m in msgs:
|
| 513 |
+
if isinstance(m, dict):
|
| 514 |
+
role = m.get("role", "")
|
| 515 |
+
content = m.get("content", "") or m.get("value", "")
|
| 516 |
+
if role in ("user", "human") and content:
|
| 517 |
+
user_msg = content
|
| 518 |
+
break
|
| 519 |
+
if not user_msg:
|
| 520 |
+
return None
|
| 521 |
+
if not calls:
|
| 522 |
+
return None
|
| 523 |
+
|
| 524 |
+
tools_str = _to_json_str(tools) if tools else ""
|
| 525 |
+
calls_str = _to_json_str(calls)
|
| 526 |
+
if not calls_str or len(calls_str) < 10:
|
| 527 |
+
return None
|
| 528 |
+
|
| 529 |
+
sys_msg = (
|
| 530 |
+
"Sen yardımcı bir asistansın. Şu araçları kullanabilirsin:\n\n"
|
| 531 |
+
f"{tools_str}\n\n"
|
| 532 |
+
"Uygun fonksiyonu çağırarak yanıtla."
|
| 533 |
+
) if tools_str else "Sen yardımcı bir asistansın. Araç çağrılarıyla yanıtla."
|
| 534 |
+
|
| 535 |
+
return {"system": sys_msg, "user": user_msg.strip(), "assistant": calls_str}
|
| 536 |
+
|
| 537 |
+
# function-calling-20k: 3 string field
|
| 538 |
+
tools = rec.get(config.get("tools_key", "tools"), "")
|
| 539 |
+
query = rec.get(config.get("query_key", "query"), "")
|
| 540 |
+
answers = rec.get(config.get("answers_key", "answers"), "")
|
| 541 |
+
if not tools or not query or not answers:
|
| 542 |
+
return None
|
| 543 |
+
if not isinstance(query, str) or len(query.strip()) < 10:
|
| 544 |
+
return None
|
| 545 |
+
sys_msg = (
|
| 546 |
+
"Sen yardımcı bir asistansın. Şu araçları kullanabilirsin:\n\n"
|
| 547 |
+
f"{tools}\n\n"
|
| 548 |
+
"Uygun fonksiyonu çağırarak yanıtla."
|
| 549 |
+
)
|
| 550 |
+
return {
|
| 551 |
+
"system": sys_msg,
|
| 552 |
+
"user": query.strip(),
|
| 553 |
+
"assistant": answers if isinstance(answers, str) else _to_json_str(answers),
|
| 554 |
+
}
|
| 555 |
+
|
| 556 |
+
|
| 557 |
+
def extract_oasst(rec: dict, config: dict) -> dict | None:
|
| 558 |
+
"""OpenAssistant tree: parent_id + role + lang + text."""
|
| 559 |
+
if rec.get("lang") != config.get("lang_filter", "tr"):
|
| 560 |
+
return None
|
| 561 |
+
# OAsst tree: ana mesaji al, cevabi al — tek-turn'lik
|
| 562 |
+
# Burada sadece "prompter" tipi root + 1 assistant cevabi alacagiz
|
| 563 |
+
role = rec.get("role", "")
|
| 564 |
+
text = rec.get("text", "")
|
| 565 |
+
if role != "prompter" or not text:
|
| 566 |
+
return None
|
| 567 |
+
# Tek-record extract zor — gercek pipeline tree icin daha karmasik
|
| 568 |
+
# Basitlestirme: prompter mesajini user yap, dummy assistant
|
| 569 |
+
# Bu yontem yetersiz olabilir — ya tum tree'yi paralel oku ya skip
|
| 570 |
+
return None # OAsst kompleks, simdilik skip
|
| 571 |
+
|
| 572 |
+
|
| 573 |
+
def extract_lima(rec: dict, config: dict) -> dict | None:
|
| 574 |
+
"""LIMA: conversations [str, str] format."""
|
| 575 |
+
convs = rec.get("conversations") or rec.get("messages")
|
| 576 |
+
if isinstance(convs, list) and len(convs) >= 2:
|
| 577 |
+
if isinstance(convs[0], str):
|
| 578 |
+
return {"user": convs[0].strip(), "assistant": convs[1].strip()}
|
| 579 |
+
return extract_messages_format(rec, config)
|
| 580 |
+
# Fallback standart alanlar
|
| 581 |
+
user = find_field(rec, config.get("input_keys", []))
|
| 582 |
+
asst = find_field(rec, config.get("output_keys", []))
|
| 583 |
+
if user and asst:
|
| 584 |
+
return {"user": user, "assistant": asst}
|
| 585 |
+
return None
|
| 586 |
+
|
| 587 |
+
|
| 588 |
+
def extract_essay(rec: dict, config: dict) -> dict | None:
|
| 589 |
+
"""Bilkent / masallar — text alanindan instruction olustur."""
|
| 590 |
+
text = find_field(rec, config.get("text_keys", []) or config.get("output_keys", []))
|
| 591 |
+
if not text:
|
| 592 |
+
# Fallback: ilk uzun string alan
|
| 593 |
+
for k, v in rec.items():
|
| 594 |
+
if isinstance(v, str) and len(v) > 200:
|
| 595 |
+
text = v
|
| 596 |
+
break
|
| 597 |
+
if not text or len(text) < 200:
|
| 598 |
+
return None
|
| 599 |
+
title = find_field(rec, ["title", "name", "baslik"])
|
| 600 |
+
|
| 601 |
+
prompts = []
|
| 602 |
+
if "masallar" in config["repo"].lower():
|
| 603 |
+
prompts = ["Bir masal anlat.", "Bana bir Türk masalı anlat.",
|
| 604 |
+
"Geleneksel bir masal yaz.", "Eski bir masal hikayesi yaz."]
|
| 605 |
+
if title:
|
| 606 |
+
prompts = [f"'{title}' masalını anlat."]
|
| 607 |
+
else:
|
| 608 |
+
prompts = ["Aşağıdaki konuda detaylı bir kompozisyon yaz.",
|
| 609 |
+
"Bu konu hakkında bir yazı yaz."]
|
| 610 |
+
prompt = random.choice(prompts)
|
| 611 |
+
if title and not any(title in p for p in prompts):
|
| 612 |
+
prompt = f"{prompt}\n\nKonu: {title}"
|
| 613 |
+
return {"user": prompt, "assistant": text[:5000]}
|
| 614 |
+
|
| 615 |
+
|
| 616 |
+
def extract_with_thinking(rec: dict, config: dict) -> dict | None:
|
| 617 |
+
"""Thinking traces ile: user + thinking + final."""
|
| 618 |
+
user = find_field(rec, config.get("input_keys", []))
|
| 619 |
+
if not user:
|
| 620 |
+
return None
|
| 621 |
+
asst = find_field(rec, config.get("output_keys", []))
|
| 622 |
+
thinking = find_field(rec, [config.get("thinking_key", "thinking")])
|
| 623 |
+
if not asst:
|
| 624 |
+
return None
|
| 625 |
+
# Eger thinking ayri alanda → ChatML icinde <think> tag'i ile birlestir
|
| 626 |
+
if thinking and thinking != asst:
|
| 627 |
+
full_asst = f"<think>\n{thinking}\n</think>\n\n{asst}"
|
| 628 |
+
else:
|
| 629 |
+
full_asst = asst
|
| 630 |
+
return {"user": user, "assistant": full_asst}
|
| 631 |
+
|
| 632 |
+
|
| 633 |
+
def extract_default(rec: dict, config: dict) -> dict | None:
|
| 634 |
+
"""Standart input/output extraction."""
|
| 635 |
+
user = find_field(rec, config.get("input_keys", []))
|
| 636 |
+
if not user:
|
| 637 |
+
return None
|
| 638 |
+
asst = find_field(rec, config.get("output_keys", []))
|
| 639 |
+
if not asst:
|
| 640 |
+
return None
|
| 641 |
+
|
| 642 |
+
if "input2_keys" in config:
|
| 643 |
+
extra = find_field(rec, config["input2_keys"])
|
| 644 |
+
if extra and len(extra) > 5 and extra not in user:
|
| 645 |
+
user = f"{user}\n\n{extra}"
|
| 646 |
+
|
| 647 |
+
sys_msg = find_field(rec, config.get("system_keys", []) or ["system"])
|
| 648 |
+
|
| 649 |
+
out = {"user": user, "assistant": asst}
|
| 650 |
+
if sys_msg:
|
| 651 |
+
out["system"] = sys_msg
|
| 652 |
+
return out
|
| 653 |
+
|
| 654 |
+
|
| 655 |
+
def extract_thinking_data(rec: dict, config: dict) -> dict | None:
|
| 656 |
+
"""ThinkingData-200K: messages list (system+user) + reasoning + answer."""
|
| 657 |
+
msgs = rec.get("messages", [])
|
| 658 |
+
reasoning = rec.get("reasoning", "")
|
| 659 |
+
answer = rec.get("answer", "")
|
| 660 |
+
if not msgs or not answer:
|
| 661 |
+
return None
|
| 662 |
+
user_msg = ""
|
| 663 |
+
sys_msg = ""
|
| 664 |
+
for m in msgs:
|
| 665 |
+
if not isinstance(m, dict):
|
| 666 |
+
continue
|
| 667 |
+
role = m.get("role", "")
|
| 668 |
+
content = m.get("content", "")
|
| 669 |
+
if role == "user":
|
| 670 |
+
user_msg = content
|
| 671 |
+
elif role == "system":
|
| 672 |
+
sys_msg = content
|
| 673 |
+
if not user_msg:
|
| 674 |
+
return None
|
| 675 |
+
# Combine reasoning + answer with <think> format
|
| 676 |
+
if reasoning and reasoning.strip():
|
| 677 |
+
full_asst = f"<think>\n{reasoning.strip()}\n</think>\n\n{answer.strip()}"
|
| 678 |
+
else:
|
| 679 |
+
full_asst = answer.strip()
|
| 680 |
+
out = {"user": user_msg, "assistant": full_asst}
|
| 681 |
+
if sys_msg:
|
| 682 |
+
out["system"] = sys_msg
|
| 683 |
+
return out
|
| 684 |
+
|
| 685 |
+
|
| 686 |
+
def extract_tot(rec: dict, config: dict) -> dict | None:
|
| 687 |
+
"""Tree of Thought: chat_format JSON string."""
|
| 688 |
+
chat_fmt = rec.get("chat_format", "")
|
| 689 |
+
if not chat_fmt:
|
| 690 |
+
# Fallback: question + judged_answer
|
| 691 |
+
q = rec.get("complex_question_tr", "")
|
| 692 |
+
ja = rec.get("judged_answer", {})
|
| 693 |
+
if isinstance(ja, dict):
|
| 694 |
+
ja_text = ja.get("final_answer", "") or ja.get("answer", "") or str(ja)
|
| 695 |
+
else:
|
| 696 |
+
ja_text = str(ja)
|
| 697 |
+
if q and ja_text:
|
| 698 |
+
return {"user": q.strip(), "assistant": ja_text.strip()}
|
| 699 |
+
return None
|
| 700 |
+
# chat_format is a JSON string with user/assistant
|
| 701 |
+
try:
|
| 702 |
+
parsed = json.loads(chat_fmt) if isinstance(chat_fmt, str) else chat_fmt
|
| 703 |
+
if isinstance(parsed, dict):
|
| 704 |
+
# Genelde {messages: [...]} veya {user: ..., assistant: ...}
|
| 705 |
+
if "messages" in parsed:
|
| 706 |
+
return extract_messages_format({"messages": parsed["messages"]}, config)
|
| 707 |
+
if "user" in parsed and "assistant" in parsed:
|
| 708 |
+
return {"user": parsed["user"], "assistant": parsed["assistant"]}
|
| 709 |
+
if isinstance(parsed, list):
|
| 710 |
+
return extract_messages_format({"messages": parsed}, config)
|
| 711 |
+
except Exception:
|
| 712 |
+
pass
|
| 713 |
+
return None
|
| 714 |
+
|
| 715 |
+
|
| 716 |
+
def extract_gpqa(rec: dict) -> dict | None:
|
| 717 |
+
"""GPQA: Question + Correct Answer + Incorrect Answer 1-3."""
|
| 718 |
+
question = rec.get("Question", "")
|
| 719 |
+
correct = rec.get("Correct Answer", "")
|
| 720 |
+
incorrects = [
|
| 721 |
+
rec.get("Incorrect Answer 1", ""),
|
| 722 |
+
rec.get("Incorrect Answer 2", ""),
|
| 723 |
+
rec.get("Incorrect Answer 3", ""),
|
| 724 |
+
]
|
| 725 |
+
incorrects = [x for x in incorrects if x]
|
| 726 |
+
if not question or not correct or len(incorrects) < 2:
|
| 727 |
+
return None
|
| 728 |
+
# Karistir
|
| 729 |
+
choices = [correct] + incorrects
|
| 730 |
+
random.shuffle(choices)
|
| 731 |
+
correct_idx = choices.index(correct)
|
| 732 |
+
correct_letter = chr(65 + correct_idx)
|
| 733 |
+
|
| 734 |
+
choices_text = "\n".join(
|
| 735 |
+
f"{chr(65+i)}) {c}" for i, c in enumerate(choices)
|
| 736 |
+
)
|
| 737 |
+
user = f"{question}\n\n{choices_text}"
|
| 738 |
+
assistant = f"Cevap: {correct_letter}) {correct}"
|
| 739 |
+
return {"user": user, "assistant": assistant}
|
| 740 |
+
|
| 741 |
+
|
| 742 |
+
def extract_quardo(rec: dict, config: dict) -> dict | None:
|
| 743 |
+
"""Quardo/Turkish-Chat_GPT-4O: prompt(system) + question(user) + chat(assistant).
|
| 744 |
+
'chat' alani genelde liste (multi-turn) ya da string olabilir."""
|
| 745 |
+
prompt = rec.get("prompt", "") or ""
|
| 746 |
+
question = rec.get("question", "") or ""
|
| 747 |
+
chat = rec.get("chat", None)
|
| 748 |
+
|
| 749 |
+
# 'chat' bazen JSON string, bazen list of dicts {role, content} olabilir
|
| 750 |
+
asst_text = ""
|
| 751 |
+
if isinstance(chat, str):
|
| 752 |
+
s = chat.strip()
|
| 753 |
+
if s.startswith("[") or s.startswith("{"):
|
| 754 |
+
try:
|
| 755 |
+
chat = json.loads(s)
|
| 756 |
+
except Exception:
|
| 757 |
+
asst_text = s
|
| 758 |
+
else:
|
| 759 |
+
asst_text = s
|
| 760 |
+
if isinstance(chat, list):
|
| 761 |
+
# Ilk assistant mesajini al
|
| 762 |
+
for m in chat:
|
| 763 |
+
if isinstance(m, dict):
|
| 764 |
+
role = m.get("role", "") or m.get("from", "")
|
| 765 |
+
content = m.get("content", "") or m.get("value", "")
|
| 766 |
+
if role in ("assistant", "gpt", "bot") and content:
|
| 767 |
+
asst_text = content
|
| 768 |
+
break
|
| 769 |
+
|
| 770 |
+
if not asst_text:
|
| 771 |
+
# Fallback: 'content' alanini dene
|
| 772 |
+
asst_text = rec.get("content", "") or ""
|
| 773 |
+
if not isinstance(asst_text, str) or len(asst_text.strip()) < 15:
|
| 774 |
+
return None
|
| 775 |
+
if not isinstance(question, str) or len(question.strip()) < 5:
|
| 776 |
+
return None
|
| 777 |
+
|
| 778 |
+
out = {"user": question.strip(), "assistant": asst_text.strip()}
|
| 779 |
+
if prompt and isinstance(prompt, str) and len(prompt.strip()) > 10:
|
| 780 |
+
out["system"] = prompt.strip()
|
| 781 |
+
return out
|
| 782 |
+
|
| 783 |
+
|
| 784 |
+
def extract_inst_format(rec: dict, config: dict) -> dict | None:
|
| 785 |
+
"""[INST] ... [/INST] formatindan parse et (Mistral/Llama2 stili)."""
|
| 786 |
+
text = rec.get("text", "")
|
| 787 |
+
if not text or "[INST]" not in text:
|
| 788 |
+
return None
|
| 789 |
+
# <s>[INST] question [/INST] answer </s>
|
| 790 |
+
m = re.search(r"\[INST\]\s*(.+?)\s*\[/INST\]\s*(.+?)\s*(?:</s>|$)",
|
| 791 |
+
text, re.DOTALL)
|
| 792 |
+
if not m:
|
| 793 |
+
return None
|
| 794 |
+
user = m.group(1).strip()
|
| 795 |
+
asst = m.group(2).strip().rstrip("</s>").strip()
|
| 796 |
+
if len(user) < 5 or len(asst) < 5:
|
| 797 |
+
return None
|
| 798 |
+
return {"user": user, "assistant": asst}
|
| 799 |
+
|
| 800 |
+
|
| 801 |
+
def extract_record(rec: dict, config: dict) -> dict | None:
|
| 802 |
+
"""Dispatcher — config'e gore dogru handler'i cagir."""
|
| 803 |
+
if config.get("is_mmlu"):
|
| 804 |
+
return extract_mmlu(rec)
|
| 805 |
+
if config.get("is_gpqa"):
|
| 806 |
+
return extract_gpqa(rec)
|
| 807 |
+
if config.get("is_function_calling"):
|
| 808 |
+
return extract_function_calling(rec, config)
|
| 809 |
+
if config.get("is_oasst"):
|
| 810 |
+
return extract_oasst(rec, config)
|
| 811 |
+
if config.get("is_lima"):
|
| 812 |
+
return extract_lima(rec, config)
|
| 813 |
+
if config.get("is_thinking_data"):
|
| 814 |
+
return extract_thinking_data(rec, config)
|
| 815 |
+
if config.get("is_tot"):
|
| 816 |
+
return extract_tot(rec, config)
|
| 817 |
+
if config.get("is_inst_format"):
|
| 818 |
+
return extract_inst_format(rec, config)
|
| 819 |
+
if config.get("is_quardo"):
|
| 820 |
+
return extract_quardo(rec, config)
|
| 821 |
+
if config.get("is_essay"):
|
| 822 |
+
return extract_essay(rec, config)
|
| 823 |
+
if config.get("is_messages"):
|
| 824 |
+
result = extract_messages_format(rec, config)
|
| 825 |
+
if result is not None:
|
| 826 |
+
return result
|
| 827 |
+
return extract_default(rec, config)
|
| 828 |
+
if config.get("thinking_key"):
|
| 829 |
+
return extract_with_thinking(rec, config)
|
| 830 |
+
return extract_default(rec, config)
|
| 831 |
+
|
| 832 |
+
|
| 833 |
+
# =====================================================================
|
| 834 |
+
# Filtreler
|
| 835 |
+
# =====================================================================
|
| 836 |
+
BAD_PATTERNS = [
|
| 837 |
+
r"https?://[^\s]+",
|
| 838 |
+
r"www\.[^\s]+\.[a-z]+",
|
| 839 |
+
r"\b\w+@\w+\.\w+\b",
|
| 840 |
+
r"Erişim tarihi[:.]?\s*\d",
|
| 841 |
+
r"Ana Sayfa\s+(Yaşam|Kadın|Spor)",
|
| 842 |
+
r"Devamını oku|Kaynak:|Reklam",
|
| 843 |
+
r"\b(KDV dahil|Ücretsiz Kargo)\b",
|
| 844 |
+
r"\[\s*reklam\s*\]",
|
| 845 |
+
r"<[^>]{1,50}>",
|
| 846 |
+
]
|
| 847 |
+
|
| 848 |
+
|
| 849 |
+
def has_repetition(text: str, n: int = 4) -> bool:
|
| 850 |
+
sentences = re.split(r"[.!?]\s+", text)
|
| 851 |
+
if len(sentences) < n:
|
| 852 |
+
return False
|
| 853 |
+
for i in range(len(sentences) - n + 1):
|
| 854 |
+
if len(set(sentences[i:i+n])) == 1 and len(sentences[i]) > 20:
|
| 855 |
+
return True
|
| 856 |
+
return False
|
| 857 |
+
|
| 858 |
+
|
| 859 |
+
def turkish_char_ratio(text: str) -> float:
|
| 860 |
+
if not text:
|
| 861 |
+
return 0.0
|
| 862 |
+
tr_chars = sum(c in "şğıöüçŞĞİÖÜÇ" for c in text)
|
| 863 |
+
letters = sum(c.isalpha() for c in text)
|
| 864 |
+
return tr_chars / max(letters, 1)
|
| 865 |
+
|
| 866 |
+
|
| 867 |
+
def is_turkish(text: str, min_ratio: float = 0.01) -> bool:
|
| 868 |
+
if len(text) < 30:
|
| 869 |
+
return True
|
| 870 |
+
tr_words = {"ve", "bir", "için", "ile", "bu", "şu", "ama", "veya",
|
| 871 |
+
"olan", "olarak", "kadar", "her", "var", "gibi", "yok",
|
| 872 |
+
"ben", "sen", "biz", "siz", "onlar", "ise", "değil",
|
| 873 |
+
"daha", "çok", "az", "diye", "ki", "de", "da"}
|
| 874 |
+
words = re.findall(r"\b\w+\b", text.lower())
|
| 875 |
+
if not words:
|
| 876 |
+
return False
|
| 877 |
+
tr_count = sum(1 for w in words if w in tr_words)
|
| 878 |
+
ratio = tr_count / len(words)
|
| 879 |
+
return ratio >= min_ratio or turkish_char_ratio(text) >= 0.02
|
| 880 |
+
|
| 881 |
+
|
| 882 |
+
def quality_check(rec: dict, is_function_calling: bool = False,
|
| 883 |
+
category: str = "general") -> tuple[bool, str]:
|
| 884 |
+
user = rec.get("user", "")
|
| 885 |
+
asst = rec.get("assistant", "")
|
| 886 |
+
|
| 887 |
+
# Kategoriye gore farkli max uzunluk limitleri
|
| 888 |
+
asst_max = {
|
| 889 |
+
"reasoning": 25000, # thinking traces uzun olabilir
|
| 890 |
+
"curated": 15000, # alican_sft uzun cevaplar
|
| 891 |
+
"knowledge": 10000,
|
| 892 |
+
"conversation": 8000,
|
| 893 |
+
"general": 6000,
|
| 894 |
+
"function": 12000,
|
| 895 |
+
}.get(category, 8000)
|
| 896 |
+
|
| 897 |
+
# Boyut
|
| 898 |
+
if len(user) < 8:
|
| 899 |
+
return False, "user_too_short"
|
| 900 |
+
if len(user) > 4000:
|
| 901 |
+
return False, "user_too_long"
|
| 902 |
+
if len(asst) < 15:
|
| 903 |
+
return False, "asst_too_short"
|
| 904 |
+
if len(asst) > asst_max:
|
| 905 |
+
return False, "asst_too_long"
|
| 906 |
+
|
| 907 |
+
# Function calling: JSON kontrolu
|
| 908 |
+
if is_function_calling:
|
| 909 |
+
s = asst.strip()
|
| 910 |
+
if not (s.startswith("[") or s.startswith("{")):
|
| 911 |
+
return False, "fc_not_json"
|
| 912 |
+
return True, "ok"
|
| 913 |
+
|
| 914 |
+
# Turkce
|
| 915 |
+
if not is_turkish(asst):
|
| 916 |
+
return False, "not_turkish"
|
| 917 |
+
|
| 918 |
+
# Kotu pattern'ler — reasoning/curated icin URL/email check'i atla
|
| 919 |
+
# (akademik citation, kod URLs olabilir; <think> tag normaldir)
|
| 920 |
+
skip_patterns_for = {"reasoning", "curated"}
|
| 921 |
+
if category not in skip_patterns_for:
|
| 922 |
+
for pat in BAD_PATTERNS:
|
| 923 |
+
if re.search(pat, asst):
|
| 924 |
+
return False, "bad_pattern"
|
| 925 |
+
else:
|
| 926 |
+
# Sadece gerçek SEO/spam pattern'leri — HTML/tag check'i YOK
|
| 927 |
+
# <think>, <|begin_of_thought|> gibi tag'ler reasoning icin normaldir
|
| 928 |
+
seo_patterns = [
|
| 929 |
+
r"Ana Sayfa\s+(Yaşam|Kadın|Spor)",
|
| 930 |
+
r"Devamını oku\s*»|Kaynak:\s*http|Reklam\s*\[",
|
| 931 |
+
r"\[\s*reklam\s*\]",
|
| 932 |
+
# Gercek HTML tag'leri (div, p, span, a, br, img, table, tr, td, h1-6)
|
| 933 |
+
r"<(?:div|p|span|a|br|img|table|tr|td|h[1-6])\b[^>]*>",
|
| 934 |
+
]
|
| 935 |
+
for pat in seo_patterns:
|
| 936 |
+
if re.search(pat, asst):
|
| 937 |
+
return False, "bad_pattern"
|
| 938 |
+
|
| 939 |
+
if has_repetition(asst):
|
| 940 |
+
return False, "repetition"
|
| 941 |
+
|
| 942 |
+
if len(asst) > 200 and turkish_char_ratio(asst) < 0.005:
|
| 943 |
+
return False, "no_turkish_chars"
|
| 944 |
+
|
| 945 |
+
if user.lower() == asst.lower():
|
| 946 |
+
return False, "echo"
|
| 947 |
+
|
| 948 |
+
# Trivial cevap (reasoning + curated hariç)
|
| 949 |
+
if category not in ("reasoning", "curated"):
|
| 950 |
+
if len(asst) < 40 and asst.count(".") <= 1 and asst.count(" ") < 5:
|
| 951 |
+
return False, "asst_trivial"
|
| 952 |
+
|
| 953 |
+
# asst_incomplete: math/code/reasoning icin gevsek kontrol
|
| 954 |
+
# Mid-word kesinti varsa kabul etme — son 30 karakterde bosluk yoksa supheli
|
| 955 |
+
if len(asst) > 50:
|
| 956 |
+
if category in ("reasoning", "conversation", "curated"):
|
| 957 |
+
# Daha gevsek: sadece mid-word kontrol
|
| 958 |
+
last30 = asst[-30:].rstrip()
|
| 959 |
+
# Eger son karakter alfabetik VE bir bosluk yoksa = kesilmis kelime
|
| 960 |
+
if (last30 and last30[-1].isalpha() and " " not in last30
|
| 961 |
+
and last30[-1] not in "ıİĞğüÜşŞçÇöÖ"):
|
| 962 |
+
# Sayilarla bitiyorsa OK (=42, =100)
|
| 963 |
+
pass
|
| 964 |
+
# Cok kisa son satir kesinti
|
| 965 |
+
if len(last30) < 3:
|
| 966 |
+
pass # OK
|
| 967 |
+
# Cogu durumda kabul et
|
| 968 |
+
else:
|
| 969 |
+
# General/knowledge/function: katı kontrol
|
| 970 |
+
if asst[-1] not in ".!?\"')]}…」、。0123456789":
|
| 971 |
+
return False, "asst_incomplete"
|
| 972 |
+
|
| 973 |
+
if asst.count("\n\n\n") > 2:
|
| 974 |
+
return False, "asst_excessive_newlines"
|
| 975 |
+
|
| 976 |
+
short_refusals = ["bilmiyorum", "yardımcı olamam", "anlamadım"]
|
| 977 |
+
if len(asst) < 60 and any(r in asst.lower() for r in short_refusals):
|
| 978 |
+
return False, "asst_refusal_too_short"
|
| 979 |
+
|
| 980 |
+
return True, "ok"
|
| 981 |
+
|
| 982 |
+
|
| 983 |
+
# =====================================================================
|
| 984 |
+
# Pipeline
|
| 985 |
+
# =====================================================================
|
| 986 |
+
def hash_text(text: str) -> str:
|
| 987 |
+
return hashlib.md5(text.lower().strip()[:500].encode()).hexdigest()
|
| 988 |
+
|
| 989 |
+
|
| 990 |
+
def process_dataset(name: str, config: dict, show_progress: bool = True,
|
| 991 |
+
use_cache: bool = True, force: bool = False):
|
| 992 |
+
cache_path = CACHE_DIR / f"{name}.jsonl"
|
| 993 |
+
cache_stats_path = CACHE_DIR / f"{name}_stats.json"
|
| 994 |
+
|
| 995 |
+
# Cache hit — onceden basariyla isleyen dataset'i tekrar indirme
|
| 996 |
+
if use_cache and not force and cache_path.exists():
|
| 997 |
+
n_lines = 0
|
| 998 |
+
accepted = []
|
| 999 |
+
try:
|
| 1000 |
+
with open(cache_path, "r", encoding="utf-8") as f:
|
| 1001 |
+
for line in f:
|
| 1002 |
+
accepted.append(json.loads(line))
|
| 1003 |
+
n_lines += 1
|
| 1004 |
+
print(f"\n{'='*60}\n[{name}] CACHE HIT — {n_lines:,} ornek "
|
| 1005 |
+
f"({cache_path.name})\n{'='*60}")
|
| 1006 |
+
stats = {"cached": True, "accepted": n_lines}
|
| 1007 |
+
if cache_stats_path.exists():
|
| 1008 |
+
with open(cache_stats_path) as sf:
|
| 1009 |
+
stats.update(json.load(sf))
|
| 1010 |
+
return accepted, stats
|
| 1011 |
+
except Exception as e:
|
| 1012 |
+
print(f" ! Cache okunamadi ({e}), yeniden indirilecek")
|
| 1013 |
+
|
| 1014 |
+
print(f"\n{'='*60}\n[{name}] {config['repo']} ({config['category']})\n{'='*60}")
|
| 1015 |
+
try:
|
| 1016 |
+
# config_name varsa subset secimi yap
|
| 1017 |
+
load_kwargs = {"split": config["split"]}
|
| 1018 |
+
if config.get("config_name"):
|
| 1019 |
+
load_kwargs["name"] = config["config_name"]
|
| 1020 |
+
ds = load_dataset(config["repo"], **load_kwargs)
|
| 1021 |
+
except Exception as e:
|
| 1022 |
+
try:
|
| 1023 |
+
# validation veya test split dene
|
| 1024 |
+
for alt_split in ["validation", "test", "train"]:
|
| 1025 |
+
if alt_split == config["split"]:
|
| 1026 |
+
continue
|
| 1027 |
+
try:
|
| 1028 |
+
load_kwargs["split"] = alt_split
|
| 1029 |
+
ds = load_dataset(config["repo"], **load_kwargs)
|
| 1030 |
+
print(f" (split degisti: {config['split']} -> {alt_split})")
|
| 1031 |
+
break
|
| 1032 |
+
except Exception:
|
| 1033 |
+
continue
|
| 1034 |
+
else:
|
| 1035 |
+
raise e
|
| 1036 |
+
except Exception as e2:
|
| 1037 |
+
print(f" ! Yuklenemedi: {e2}")
|
| 1038 |
+
return [], {"error": str(e2)}
|
| 1039 |
+
|
| 1040 |
+
total = len(ds)
|
| 1041 |
+
target = config.get("target", total)
|
| 1042 |
+
print(f" Toplam: {total:,} satir | hedef: {target:,}")
|
| 1043 |
+
|
| 1044 |
+
accepted = []
|
| 1045 |
+
reject_reasons = Counter()
|
| 1046 |
+
rejected_samples = []
|
| 1047 |
+
seen_hashes = set()
|
| 1048 |
+
is_fc = config.get("is_function_calling", False)
|
| 1049 |
+
cat = config["category"]
|
| 1050 |
+
|
| 1051 |
+
iterator = ds if not show_progress else tqdm(ds, total=total, desc=f" {name}")
|
| 1052 |
+
for rec in iterator:
|
| 1053 |
+
# Erken durma — target'a ulaştık
|
| 1054 |
+
if len(accepted) >= target * 1.5: # %50 fazla topla, sonra sample
|
| 1055 |
+
break
|
| 1056 |
+
|
| 1057 |
+
extracted = extract_record(rec, config)
|
| 1058 |
+
if extracted is None:
|
| 1059 |
+
reject_reasons["no_fields"] += 1
|
| 1060 |
+
continue
|
| 1061 |
+
|
| 1062 |
+
# Source-spesifik ek filtreler (config'ten gelir)
|
| 1063 |
+
if config.get("skip_translation"):
|
| 1064 |
+
uq = extracted.get("user", "")
|
| 1065 |
+
if any(kw in uq for kw in ("çevirin", "çevir ", "çeviriniz",
|
| 1066 |
+
"tercüme", "Translate", "translate",
|
| 1067 |
+
"limerik")):
|
| 1068 |
+
reject_reasons["translation_filtered"] += 1
|
| 1069 |
+
continue
|
| 1070 |
+
min_alen = config.get("min_asst_len")
|
| 1071 |
+
if min_alen and len(extracted.get("assistant", "")) < min_alen:
|
| 1072 |
+
reject_reasons["below_min_asst_len"] += 1
|
| 1073 |
+
continue
|
| 1074 |
+
|
| 1075 |
+
ok, reason = quality_check(extracted, is_function_calling=is_fc, category=cat)
|
| 1076 |
+
if not ok:
|
| 1077 |
+
reject_reasons[reason] += 1
|
| 1078 |
+
if len(rejected_samples) < 3:
|
| 1079 |
+
rejected_samples.append({
|
| 1080 |
+
"reason": reason,
|
| 1081 |
+
"user": extracted.get("user", "")[:200],
|
| 1082 |
+
"assistant": extracted.get("assistant", "")[:200],
|
| 1083 |
+
})
|
| 1084 |
+
continue
|
| 1085 |
+
|
| 1086 |
+
h = hash_text(extracted["user"] + extracted["assistant"][:200])
|
| 1087 |
+
if h in seen_hashes:
|
| 1088 |
+
reject_reasons["duplicate"] += 1
|
| 1089 |
+
continue
|
| 1090 |
+
seen_hashes.add(h)
|
| 1091 |
+
|
| 1092 |
+
extracted["source"] = name
|
| 1093 |
+
extracted["category"] = cat
|
| 1094 |
+
if config.get("tag"):
|
| 1095 |
+
extracted["tag"] = config["tag"]
|
| 1096 |
+
accepted.append(extracted)
|
| 1097 |
+
|
| 1098 |
+
# Target'a indir (random sample)
|
| 1099 |
+
if len(accepted) > target:
|
| 1100 |
+
random.seed(42)
|
| 1101 |
+
accepted = random.sample(accepted, target)
|
| 1102 |
+
|
| 1103 |
+
stats = {
|
| 1104 |
+
"repo": config["repo"],
|
| 1105 |
+
"category": cat,
|
| 1106 |
+
"total_loaded": total,
|
| 1107 |
+
"accepted": len(accepted),
|
| 1108 |
+
"accept_rate": len(accepted) / max(total, 1),
|
| 1109 |
+
"target": target,
|
| 1110 |
+
"rejects": dict(reject_reasons.most_common(10)),
|
| 1111 |
+
"rejected_samples": rejected_samples,
|
| 1112 |
+
}
|
| 1113 |
+
print(f" Accept: {len(accepted):,} (target {target}) | "
|
| 1114 |
+
f"raw {sum(reject_reasons.values())} reject")
|
| 1115 |
+
top3 = reject_reasons.most_common(3)
|
| 1116 |
+
print(f" Top rejects: {top3}")
|
| 1117 |
+
|
| 1118 |
+
# Cache'e yaz — sadece basarili (>0 ornek) sonuclari
|
| 1119 |
+
if len(accepted) > 0:
|
| 1120 |
+
try:
|
| 1121 |
+
with open(cache_path, "w", encoding="utf-8") as f:
|
| 1122 |
+
for r in accepted:
|
| 1123 |
+
f.write(json.dumps(r, ensure_ascii=False) + "\n")
|
| 1124 |
+
with open(cache_stats_path, "w", encoding="utf-8") as sf:
|
| 1125 |
+
json.dump(stats, sf, indent=2, ensure_ascii=False)
|
| 1126 |
+
print(f" [cache] {cache_path.name} yazildi")
|
| 1127 |
+
except Exception as e:
|
| 1128 |
+
print(f" ! Cache yazma hatasi: {e}")
|
| 1129 |
+
return accepted, stats
|
| 1130 |
+
|
| 1131 |
+
|
| 1132 |
+
def to_chatml(rec: dict) -> dict:
|
| 1133 |
+
msgs = []
|
| 1134 |
+
if rec.get("system"):
|
| 1135 |
+
msgs.append({"role": "system", "content": rec["system"]})
|
| 1136 |
+
msgs.append({"role": "user", "content": rec["user"]})
|
| 1137 |
+
msgs.append({"role": "assistant", "content": rec["assistant"]})
|
| 1138 |
+
return {
|
| 1139 |
+
"messages": msgs,
|
| 1140 |
+
"source": rec.get("source", "unknown"),
|
| 1141 |
+
"category": rec.get("category", "general"),
|
| 1142 |
+
"tag": rec.get("tag", ""),
|
| 1143 |
+
}
|
| 1144 |
+
|
| 1145 |
+
|
| 1146 |
+
def main():
|
| 1147 |
+
parser = argparse.ArgumentParser()
|
| 1148 |
+
parser.add_argument("--datasets", nargs="+", default=None,
|
| 1149 |
+
help="Sadece secili datasets (varsayilan: hepsi)")
|
| 1150 |
+
parser.add_argument("--categories", nargs="+", default=None,
|
| 1151 |
+
choices=CATEGORIES,
|
| 1152 |
+
help="Sadece secili kategoriler")
|
| 1153 |
+
parser.add_argument("--target", type=int, default=150000,
|
| 1154 |
+
help="Hedef TOPLAM ornek sayisi (varsayilan: 150K)")
|
| 1155 |
+
parser.add_argument("--out", type=str,
|
| 1156 |
+
default=str(DATA_DIR / "01_collected.jsonl"))
|
| 1157 |
+
parser.add_argument("--no-progress", action="store_true")
|
| 1158 |
+
parser.add_argument("--skip-failed", action="store_true",
|
| 1159 |
+
help="Yuklenemeyen datasets'i sessizce atla")
|
| 1160 |
+
parser.add_argument("--force", nargs="*", default=None,
|
| 1161 |
+
help="Bu dataset'leri cache'e ragmen yeniden indir "
|
| 1162 |
+
"(boş list → tumu, isim list → sadece bunlar)")
|
| 1163 |
+
parser.add_argument("--clear-cache", action="store_true",
|
| 1164 |
+
help="Cache'i tamamen sil ve sifirdan basla")
|
| 1165 |
+
parser.add_argument("--no-cache", action="store_true",
|
| 1166 |
+
help="Cache kullanma, her seferinde yeniden indir")
|
| 1167 |
+
args = parser.parse_args()
|
| 1168 |
+
|
| 1169 |
+
if args.clear_cache:
|
| 1170 |
+
import shutil
|
| 1171 |
+
if CACHE_DIR.exists():
|
| 1172 |
+
shutil.rmtree(CACHE_DIR)
|
| 1173 |
+
CACHE_DIR.mkdir(parents=True, exist_ok=True)
|
| 1174 |
+
print(f"Cache temizlendi: {CACHE_DIR}")
|
| 1175 |
+
|
| 1176 |
+
random.seed(42)
|
| 1177 |
+
|
| 1178 |
+
# Dataset secimi
|
| 1179 |
+
if args.datasets:
|
| 1180 |
+
selected = {n: c for n, c in DATASETS.items() if n in args.datasets}
|
| 1181 |
+
elif args.categories:
|
| 1182 |
+
selected = {n: c for n, c in DATASETS.items()
|
| 1183 |
+
if c["category"] in args.categories}
|
| 1184 |
+
else:
|
| 1185 |
+
selected = DATASETS
|
| 1186 |
+
|
| 1187 |
+
print(f"Toplanacak datasets: {len(selected)}")
|
| 1188 |
+
print(f"Kategoriler: {Counter(c['category'] for c in selected.values())}")
|
| 1189 |
+
|
| 1190 |
+
all_records = []
|
| 1191 |
+
all_stats = {}
|
| 1192 |
+
|
| 1193 |
+
# Force list — boş list → hepsini, name list → sadece bunları
|
| 1194 |
+
force_set = None
|
| 1195 |
+
if args.force is not None:
|
| 1196 |
+
force_set = set(args.force) if args.force else set(selected.keys())
|
| 1197 |
+
|
| 1198 |
+
for name, config in selected.items():
|
| 1199 |
+
force_this = force_set is not None and name in force_set
|
| 1200 |
+
records, stats = process_dataset(
|
| 1201 |
+
name, config,
|
| 1202 |
+
show_progress=not args.no_progress,
|
| 1203 |
+
use_cache=not args.no_cache,
|
| 1204 |
+
force=force_this,
|
| 1205 |
+
)
|
| 1206 |
+
all_records.extend(records)
|
| 1207 |
+
all_stats[name] = stats
|
| 1208 |
+
if "error" in stats and not args.skip_failed:
|
| 1209 |
+
print(f" ! {name} yuklenemedi — devam edilecek")
|
| 1210 |
+
|
| 1211 |
+
# Cross-dataset dedup
|
| 1212 |
+
print(f"\n{'='*60}\nCROSS-DATASET DEDUP\n{'='*60}")
|
| 1213 |
+
seen = set()
|
| 1214 |
+
deduped = []
|
| 1215 |
+
for rec in all_records:
|
| 1216 |
+
h = hash_text(rec["user"] + rec["assistant"][:200])
|
| 1217 |
+
if h not in seen:
|
| 1218 |
+
seen.add(h)
|
| 1219 |
+
deduped.append(rec)
|
| 1220 |
+
removed = len(all_records) - len(deduped)
|
| 1221 |
+
print(f" Cross-dataset dedup: {removed:,} silindi")
|
| 1222 |
+
all_records = deduped
|
| 1223 |
+
|
| 1224 |
+
# Kategori bazli ozet (final shuffle once)
|
| 1225 |
+
cat_counts_before = Counter(r["category"] for r in all_records)
|
| 1226 |
+
print(f"\nKategori bazli (filter sonrasi):")
|
| 1227 |
+
for cat, n in cat_counts_before.most_common():
|
| 1228 |
+
print(f" {cat:<15} {n:>7,}")
|
| 1229 |
+
|
| 1230 |
+
# Target'a uydur (stratified by category)
|
| 1231 |
+
if args.target and len(all_records) > args.target:
|
| 1232 |
+
# Target kategori oranlarini koru
|
| 1233 |
+
# mmlu gated, oasst skip → ayarlı oranlar:
|
| 1234 |
+
# general(40) + reasoning(50) + knowledge(15) + conversation(20) + curated(10) + function(5) ≈ 140
|
| 1235 |
+
cat_target = {
|
| 1236 |
+
"general": int(args.target * 40 / 140),
|
| 1237 |
+
"reasoning": int(args.target * 50 / 140),
|
| 1238 |
+
"knowledge": int(args.target * 15 / 140),
|
| 1239 |
+
"conversation": int(args.target * 20 / 140),
|
| 1240 |
+
"curated": int(args.target * 10 / 140),
|
| 1241 |
+
"function": int(args.target * 5 / 140),
|
| 1242 |
+
}
|
| 1243 |
+
# Eger args.target != 150K, oranlar 150K bazli, normalizasyon
|
| 1244 |
+
by_cat = defaultdict(list)
|
| 1245 |
+
for r in all_records:
|
| 1246 |
+
by_cat[r["category"]].append(r)
|
| 1247 |
+
sampled = []
|
| 1248 |
+
for cat, recs in by_cat.items():
|
| 1249 |
+
random.shuffle(recs)
|
| 1250 |
+
n = min(cat_target.get(cat, len(recs)), len(recs))
|
| 1251 |
+
sampled.extend(recs[:n])
|
| 1252 |
+
all_records = sampled
|
| 1253 |
+
print(f"\n Stratified downsample: {len(all_records):,}")
|
| 1254 |
+
|
| 1255 |
+
random.shuffle(all_records)
|
| 1256 |
+
|
| 1257 |
+
# Yaz
|
| 1258 |
+
out_path = Path(args.out)
|
| 1259 |
+
out_path.parent.mkdir(parents=True, exist_ok=True)
|
| 1260 |
+
with open(out_path, "w", encoding="utf-8") as f:
|
| 1261 |
+
for rec in all_records:
|
| 1262 |
+
f.write(json.dumps(to_chatml(rec), ensure_ascii=False) + "\n")
|
| 1263 |
+
print(f"\n[OK] Yazildi: {out_path}")
|
| 1264 |
+
print(f" {len(all_records):,} ornek")
|
| 1265 |
+
|
| 1266 |
+
# Source + category dagilim
|
| 1267 |
+
src_counts = Counter(r["source"] for r in all_records)
|
| 1268 |
+
cat_counts = Counter(r["category"] for r in all_records)
|
| 1269 |
+
print(f"\n Kategori dagilim:")
|
| 1270 |
+
for cat, n in cat_counts.most_common():
|
| 1271 |
+
print(f" {cat:<15} {n:>7,}")
|
| 1272 |
+
print(f"\n Source dagilim:")
|
| 1273 |
+
for src, n in src_counts.most_common(30):
|
| 1274 |
+
print(f" {src:<25} {n:>7,}")
|
| 1275 |
+
|
| 1276 |
+
# Stats kaydet
|
| 1277 |
+
stats_summary = {
|
| 1278 |
+
"total_collected": len(all_records),
|
| 1279 |
+
"by_category": dict(cat_counts),
|
| 1280 |
+
"by_source": dict(src_counts),
|
| 1281 |
+
"datasets": all_stats,
|
| 1282 |
+
}
|
| 1283 |
+
stats_path = out_path.with_name(out_path.stem.replace("collected", "stats") + ".json")
|
| 1284 |
+
with open(stats_path, "w", encoding="utf-8") as f:
|
| 1285 |
+
json.dump(stats_summary, f, indent=2, ensure_ascii=False)
|
| 1286 |
+
print(f" Stats: {stats_path}")
|
| 1287 |
+
|
| 1288 |
+
# Rejected ornekler
|
| 1289 |
+
rejected_path = out_path.with_name(out_path.stem.replace("collected", "rejected_samples") + ".jsonl")
|
| 1290 |
+
with open(rejected_path, "w", encoding="utf-8") as f:
|
| 1291 |
+
for name, stats in all_stats.items():
|
| 1292 |
+
for sample in stats.get("rejected_samples", []):
|
| 1293 |
+
sample["dataset"] = name
|
| 1294 |
+
f.write(json.dumps(sample, ensure_ascii=False) + "\n")
|
| 1295 |
+
print(f" Rejected sample inceleme: {rejected_path}")
|
| 1296 |
+
|
| 1297 |
+
print(f"\nSonraki adim:")
|
| 1298 |
+
print(f" python sft_02_finalize.py (kalite kontrol + ChatML push)")
|
| 1299 |
+
|
| 1300 |
+
|
| 1301 |
+
if __name__ == "__main__":
|
| 1302 |
+
main()
|