musabc commited on
Commit
9bdf3de
·
verified ·
1 Parent(s): a5e6dd6

Upload sft_01_collect_hf_datasets.py with huggingface_hub

Browse files
Files changed (1) hide show
  1. sft_01_collect_hf_datasets.py +1302 -0
sft_01_collect_hf_datasets.py ADDED
@@ -0,0 +1,1302 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ """
2
+ V5-SFT Asama 1: HuggingFace Turkce instruction datasetlerini topla + filtrele.
3
+
4
+ 150K hedef, 25+ dataset, 6 kategori:
5
+ - General Instruction (40K)
6
+ - Reasoning (45K)
7
+ - Knowledge/QA (30K)
8
+ - Conversation & Cultural (20K)
9
+ - Quality Curated (10K)
10
+ - Function Calling (5K)
11
+
12
+ Cikti:
13
+ data/sft/01_collected.jsonl (filter sonrasi, ChatML format)
14
+ data/sft/01_stats.json (her dataset icin istatistik)
15
+ data/sft/01_rejected_samples.jsonl (incelemek icin)
16
+
17
+ Kullanim:
18
+ python sft_01_collect_hf_datasets.py # 150K hedef
19
+ python sft_01_collect_hf_datasets.py --target 50000 # daha az
20
+ python sft_01_collect_hf_datasets.py --categories reasoning # tek kategori
21
+ python sft_01_collect_hf_datasets.py --datasets turkish_mmlu # tek dataset
22
+ python sft_01_collect_hf_datasets.py --skip-failed # erisilemeyenleri atla
23
+ """
24
+
25
+ import argparse
26
+ import hashlib
27
+ import json
28
+ import os
29
+ import random
30
+ import re
31
+ import sys
32
+ import warnings
33
+ from collections import Counter, defaultdict
34
+ from pathlib import Path
35
+
36
+ os.environ["HF_HUB_DISABLE_SYMLINKS_WARNING"] = "1"
37
+ warnings.filterwarnings("ignore", category=UserWarning, module="huggingface_hub")
38
+
39
+ try:
40
+ from datasets import load_dataset, get_dataset_config_names
41
+ from tqdm import tqdm
42
+ except ImportError:
43
+ print("! Bagimliliklar eksik. Yukle:")
44
+ print(" pip install datasets tqdm")
45
+ sys.exit(1)
46
+
47
+ DATA_DIR = Path(__file__).parent / "data" / "sft"
48
+ CACHE_DIR = DATA_DIR / "cache"
49
+ DATA_DIR.mkdir(parents=True, exist_ok=True)
50
+ CACHE_DIR.mkdir(parents=True, exist_ok=True)
51
+
52
+ # =====================================================================
53
+ # Dataset Konfigurasyonlari (Kategori bazli)
54
+ # =====================================================================
55
+ DATASETS = {
56
+ # ========== GENERAL INSTRUCTION (40K) ==========
57
+ # instructurca DROPPED — kalitesiz (broken Turkce-translated code, weird prompts).
58
+ # Yerine 4 yuksek kaliteli kaynak (toplam 25K):
59
+ "fineweb_augmented_tr": {
60
+ "repo": "Ba2han/UltraFineWeb-Augmented_TR_Instruct",
61
+ "split": "train",
62
+ "category": "general",
63
+ "target": 14000, # eksik 4K'yi buradan kapat (kaliteli kaynak)
64
+ "is_messages": True,
65
+ },
66
+ "quardo_gpt4o_tr": {
67
+ "repo": "Quardo/Turkish-Chat_GPT-4O",
68
+ "split": "train",
69
+ "category": "general",
70
+ "target": 6000,
71
+ "is_quardo": True,
72
+ },
73
+ "alpaca_evol_tr": {
74
+ "repo": "malhajar/alpaca-evol-instruct-turkish",
75
+ "split": "train",
76
+ "category": "general",
77
+ "target": 5000,
78
+ "input_keys": ["evolved_instruct", "instruction"],
79
+ "output_keys": ["response", "output"],
80
+ },
81
+ "gpteacher_tr": {
82
+ "repo": "umarigan/GPTeacher-General-Instruct-tr",
83
+ "split": "train",
84
+ "category": "general",
85
+ "target": 3000,
86
+ "input_keys": ["instruction"],
87
+ "input2_keys": ["input"],
88
+ "output_keys": ["response", "output"],
89
+ # Ceviri gorevlerinde geri-ceviri hatasi var, asst min uzunluk yukseltildi
90
+ "min_asst_len": 100,
91
+ "skip_translation": True,
92
+ },
93
+ "alpaca_gpt4": {
94
+ "repo": "malhajar/alpaca-gpt4-tr",
95
+ "split": "train",
96
+ "category": "general",
97
+ "target": 10000,
98
+ # Turkish fields have -turkish suffix
99
+ "input_keys": ["instruction-turkish"],
100
+ "input2_keys": ["input-turkish"],
101
+ "output_keys": ["output-turkish"],
102
+ },
103
+ "dolly_15k_tr": {
104
+ "repo": "atasoglu/databricks-dolly-15k-tr",
105
+ "split": "train",
106
+ "category": "general",
107
+ "target": 5000,
108
+ "input_keys": ["instruction", "soru"],
109
+ "input2_keys": ["context", "input"],
110
+ "output_keys": ["response", "output", "cevap"],
111
+ },
112
+ "no_robots_tr": {
113
+ "repo": "beratcmn/no_robots_turkish",
114
+ "split": "train",
115
+ "category": "general",
116
+ "target": 5000,
117
+ # Translation fields
118
+ "input_keys": ["question_translation"],
119
+ "output_keys": ["answer_translation"],
120
+ },
121
+
122
+ # ========== REASONING (45K) ⭐ ==========
123
+ "thinking_data_200k": {
124
+ "repo": "erythropygia/ThinkingData-200K-Turkish",
125
+ "split": "train",
126
+ "category": "reasoning",
127
+ "target": 18000,
128
+ # Schema: messages list (system+user) + reasoning + answer
129
+ "is_thinking_data": True,
130
+ },
131
+ "metamathqa_tr": {
132
+ "repo": "onur48/MetaMathQA-Turkish-corrected",
133
+ "split": "train",
134
+ "category": "reasoning",
135
+ "target": 12000,
136
+ "input_keys": ["query", "question", "problem", "input"],
137
+ "output_keys": ["response", "output", "answer"],
138
+ },
139
+ "gsm8k_tr": {
140
+ "repo": "ytu-ce-cosmos/gsm8k_tr",
141
+ "split": "train",
142
+ "category": "reasoning",
143
+ "target": 8000,
144
+ "input_keys": ["question", "input", "soru"],
145
+ "output_keys": ["answer", "output", "cevap"],
146
+ },
147
+ "openthoughts_tr": {
148
+ "repo": "selimc/OpenThoughts-TR-18k",
149
+ "split": "train",
150
+ "category": "reasoning",
151
+ "target": 4000,
152
+ "input_keys": ["prompt"],
153
+ "output_keys": ["response"],
154
+ },
155
+ "turkish_medical_reasoning": {
156
+ "repo": "ituperceptron/turkish_medical_reasoning",
157
+ "split": "train",
158
+ "category": "reasoning",
159
+ "target": 2000,
160
+ "input_keys": ["question"],
161
+ "output_keys": ["answer_content"],
162
+ "thinking_key": "reasoning_content", # ayri thinking alani
163
+ },
164
+ "tree_of_thought_tr": {
165
+ "repo": "emre/ct_tree_of_thought_turkish",
166
+ "split": "train",
167
+ "category": "reasoning",
168
+ "target": 600,
169
+ # Special handler — chat_format alani var (JSON)
170
+ "is_tot": True,
171
+ },
172
+ "medium_math_tr": {
173
+ "repo": "erayalp/medium_turkish_math_reasoning",
174
+ "split": "train",
175
+ "category": "reasoning",
176
+ "target": 1000,
177
+ "input_keys": ["question", "problem", "instruction"],
178
+ "output_keys": ["answer", "solution", "response"],
179
+ },
180
+
181
+ # ========== KNOWLEDGE / QA (30K) ⭐ ==========
182
+ # turkish_mmlu gated — skip (kullanici elle erisim isteyebilir)
183
+ "turkish_exam": {
184
+ "repo": "bezir/turkish_exam_instructions",
185
+ "split": "train",
186
+ "category": "knowledge",
187
+ "target": 6000,
188
+ "input_keys": ["instruction", "soru", "question", "prompt"],
189
+ "output_keys": ["output", "response", "cevap", "answer"],
190
+ },
191
+ "wikirag_tr": {
192
+ "repo": "Metin/WikiRAG-TR",
193
+ "split": "train",
194
+ "category": "knowledge",
195
+ "target": 5000,
196
+ "input_keys": ["soru", "question", "instruction"],
197
+ "input2_keys": ["context", "passage"],
198
+ "output_keys": ["cevap", "answer", "response", "output"],
199
+ },
200
+ "instruct_papers_tr": {
201
+ "repo": "selimc/InstructPapers-TR",
202
+ "split": "train",
203
+ "category": "knowledge",
204
+ "target": 2000,
205
+ "input_keys": ["instruction", "question", "soru"],
206
+ "input2_keys": ["context", "abstract"],
207
+ "output_keys": ["output", "response", "answer"],
208
+ },
209
+ "truthful_qa_tr": {
210
+ "repo": "malhajar/truthfull_qa-tr",
211
+ "config_name": "generation", # config secimi gerekli
212
+ "split": "validation", # truthful_qa'da train yok, validation var
213
+ "category": "knowledge",
214
+ "target": 800,
215
+ "input_keys": ["question", "soru"],
216
+ "output_keys": ["best_answer", "correct_answers", "answer"],
217
+ },
218
+ "gpqa_tr": {
219
+ "repo": "ytu-ce-cosmos/gpqa-extended_tr",
220
+ "split": "train",
221
+ "category": "knowledge",
222
+ "target": 500,
223
+ # Special schema: Question + Correct Answer + Incorrect Answers 1-3
224
+ "is_gpqa": True,
225
+ },
226
+ "finance_qa": {
227
+ "repo": "umarigan/turkiye_finance_qa",
228
+ "split": "train",
229
+ "category": "knowledge",
230
+ "target": 400,
231
+ "input_keys": ["question", "soru", "instruction"],
232
+ "output_keys": ["answer", "cevap", "response", "output"],
233
+ },
234
+
235
+ # ========== CONVERSATION & CULTURAL (20K) ⭐ ==========
236
+ # oasst1_tr — tree format kompleks, skip (TR sample az)
237
+ "turkish_poems": {
238
+ "repo": "beratcmn/instruction-turkish-poems",
239
+ "split": "train",
240
+ "category": "conversation",
241
+ "target": 4000,
242
+ "input_keys": ["instruction"],
243
+ "output_keys": ["poem"],
244
+ },
245
+ "turkish_recipes": {
246
+ "repo": "mertbozkurt/llama2-TR-recipe",
247
+ "split": "train",
248
+ "category": "conversation",
249
+ "target": 4000,
250
+ # Special: text field has [INST]...[/INST] format
251
+ "is_inst_format": True,
252
+ },
253
+ "everyday_conv_tr": {
254
+ "repo": "SoAp9035/everyday-conversations-tur",
255
+ "split": "train",
256
+ "category": "conversation",
257
+ "target": 2000,
258
+ "input_keys": ["messages", "instruction", "prompt"],
259
+ "output_keys": ["response", "output"],
260
+ "is_messages": True,
261
+ },
262
+ "soap_instructions": {
263
+ "repo": "SoAp9035/turkish_instructions",
264
+ "split": "train",
265
+ "category": "conversation",
266
+ "target": 1000,
267
+ "input_keys": ["user"],
268
+ "output_keys": ["assistant"],
269
+ },
270
+ "masallar": {
271
+ "repo": "umutphp/masallar",
272
+ "split": "train",
273
+ "category": "conversation",
274
+ "target": 1000,
275
+ "text_keys": ["text", "story", "content"],
276
+ "is_essay": True,
277
+ },
278
+
279
+ # ========== QUALITY CURATED (10K) ==========
280
+ "aya_tr": {
281
+ "repo": "sayhan/aya_dataset_tur",
282
+ "split": "train",
283
+ "category": "curated",
284
+ "target": 4000,
285
+ "input_keys": ["inputs", "instruction", "prompt"],
286
+ "output_keys": ["targets", "output", "response"],
287
+ },
288
+ "alican_sft": {
289
+ "repo": "AlicanKiraz0/Turkish-SFT-Dataset-v1.0",
290
+ "split": "train",
291
+ "category": "curated",
292
+ "target": 4000,
293
+ # Direct ChatML: system/user/assistant
294
+ "input_keys": ["user"],
295
+ "output_keys": ["assistant"],
296
+ "system_keys": ["system"],
297
+ },
298
+ "lima_tr": {
299
+ "repo": "beratcmn/lima-tr",
300
+ "split": "train",
301
+ "category": "curated",
302
+ "target": 1500,
303
+ # Translated fields
304
+ "input_keys": ["translated"],
305
+ "output_keys": ["translated_reply"],
306
+ },
307
+ "general_knowledge_qa": {
308
+ "repo": "nisancoskun/turkish_general_knowledge_qa",
309
+ "split": "train",
310
+ "category": "curated",
311
+ "target": 60,
312
+ "input_keys": ["question", "soru", "instruction"],
313
+ "output_keys": ["answer", "cevap", "response"],
314
+ },
315
+
316
+ # ========== FUNCTION CALLING (5K) ==========
317
+ "function_calling": {
318
+ "repo": "atasoglu/turkish-function-calling-20k",
319
+ "split": "train",
320
+ "category": "function",
321
+ "target": 3000,
322
+ "is_function_calling": True,
323
+ "tools_key": "tools",
324
+ "query_key": "query",
325
+ "answers_key": "answers",
326
+ "tag": "function_calling",
327
+ },
328
+ "tool_calling": {
329
+ "repo": "atasoglu/turkish-tool-calling-10k",
330
+ "split": "train",
331
+ "category": "function",
332
+ "target": 2000,
333
+ "is_function_calling": True,
334
+ "is_tool_calling": True,
335
+ "tag": "tool_calling",
336
+ },
337
+ }
338
+
339
+ CATEGORIES = ["general", "reasoning", "knowledge", "conversation", "curated", "function"]
340
+
341
+
342
+ # =====================================================================
343
+ # Helper Fonksiyonlar
344
+ # =====================================================================
345
+ def find_field(record: dict, candidates: list) -> str:
346
+ """Recordda alan ara — basta/sonda bosluga tolerans."""
347
+ stripped_keys = {k.strip(): k for k in record.keys()}
348
+ for c in candidates:
349
+ c_strip = c.strip()
350
+ if c_strip in stripped_keys:
351
+ val = record[stripped_keys[c_strip]]
352
+ if isinstance(val, str) and val.strip():
353
+ return val.strip()
354
+ if isinstance(val, list) and val:
355
+ # Listede ilk string item al
356
+ for item in val:
357
+ if isinstance(item, str) and item.strip():
358
+ return item.strip()
359
+ return ""
360
+
361
+
362
+ def find_field_any(record: dict, candidates: list):
363
+ """find_field ama list/dict de dondurur."""
364
+ stripped_keys = {k.strip(): k for k in record.keys()}
365
+ for c in candidates:
366
+ c_strip = c.strip()
367
+ if c_strip in stripped_keys:
368
+ val = record[stripped_keys[c_strip]]
369
+ if val:
370
+ return val
371
+ return None
372
+
373
+
374
+ # =====================================================================
375
+ # Extract — Source-specific handlers
376
+ # =====================================================================
377
+ def extract_messages_format(rec: dict, config: dict) -> dict | None:
378
+ """messages: [{role, content}, ...] format."""
379
+ msgs = find_field_any(rec, ["messages", "conversations"])
380
+ if not isinstance(msgs, list) or len(msgs) < 2:
381
+ return None
382
+ user_msg = None
383
+ asst_msg = None
384
+ sys_msg = None
385
+ for m in msgs:
386
+ if not isinstance(m, dict):
387
+ continue
388
+ role = m.get("role") or m.get("from", "")
389
+ content = m.get("content") or m.get("value", "")
390
+ if not content:
391
+ continue
392
+ if role == "system":
393
+ sys_msg = content
394
+ elif role in ("user", "human"):
395
+ user_msg = content
396
+ elif role in ("assistant", "gpt", "bot"):
397
+ asst_msg = content
398
+ if not user_msg or not asst_msg:
399
+ return None
400
+ out = {"user": user_msg, "assistant": asst_msg}
401
+ if sys_msg:
402
+ out["system"] = sys_msg
403
+ return out
404
+
405
+
406
+ def extract_mmlu(rec: dict) -> dict | None:
407
+ """MMLU MCQA: question + choices + answer."""
408
+ question = (rec.get("question") or rec.get("soru") or
409
+ rec.get("Soru") or rec.get("Question") or "")
410
+ if not question or len(question) < 10:
411
+ return None
412
+
413
+ # Choices farkli isimler
414
+ choices = None
415
+ for key in ["choices", "options", "secenekler", "Choices"]:
416
+ if key in rec:
417
+ choices = rec[key]
418
+ break
419
+ if choices is None:
420
+ # A/B/C/D ayri alanlar
421
+ opts = []
422
+ for letter in ["A", "B", "C", "D", "E"]:
423
+ for k in [letter, f"choice_{letter}", f"option_{letter}",
424
+ f"secenek_{letter}", letter.lower()]:
425
+ if k in rec and rec[k]:
426
+ opts.append(str(rec[k]))
427
+ break
428
+ if opts:
429
+ choices = opts
430
+
431
+ if not choices:
432
+ return None
433
+ if isinstance(choices, str):
434
+ # Bazi datasetlerde virgulle ayrilmis
435
+ choices = [c.strip() for c in choices.split("|") if c.strip()]
436
+
437
+ # Cevap — letter or index
438
+ answer = (rec.get("answer") or rec.get("correct") or
439
+ rec.get("cevap") or rec.get("Answer") or "")
440
+
441
+ if isinstance(answer, int):
442
+ if 0 <= answer < len(choices):
443
+ answer_text = choices[answer]
444
+ answer_letter = chr(65 + answer)
445
+ else:
446
+ return None
447
+ elif isinstance(answer, str):
448
+ answer = answer.strip()
449
+ if len(answer) == 1 and answer.upper() in "ABCDE":
450
+ idx = ord(answer.upper()) - 65
451
+ if idx < len(choices):
452
+ answer_text = choices[idx]
453
+ answer_letter = answer.upper()
454
+ else:
455
+ return None
456
+ else:
457
+ # Cevap dogrudan metin
458
+ answer_text = answer
459
+ answer_letter = "?"
460
+ else:
461
+ return None
462
+
463
+ # Format
464
+ choices_text = "\n".join(
465
+ f"{chr(65+i)}) {c}" for i, c in enumerate(choices)
466
+ )
467
+ user = f"{question}\n\n{choices_text}"
468
+ assistant = f"Cevap: {answer_letter}) {answer_text}"
469
+ return {"user": user, "assistant": assistant}
470
+
471
+
472
+ def _parse_maybe_json(val):
473
+ """List/dict/string — hepsini handle et."""
474
+ if val is None:
475
+ return None
476
+ if isinstance(val, (list, dict)):
477
+ return val
478
+ if isinstance(val, str):
479
+ s = val.strip()
480
+ if s.startswith(("[", "{")):
481
+ try:
482
+ return json.loads(s)
483
+ except Exception:
484
+ return s
485
+ return s
486
+ return val
487
+
488
+
489
+ def _to_json_str(val):
490
+ """Tum tipleri JSON string'e cevir."""
491
+ if val is None:
492
+ return ""
493
+ if isinstance(val, str):
494
+ return val
495
+ try:
496
+ return json.dumps(val, ensure_ascii=False)
497
+ except Exception:
498
+ return str(val)
499
+
500
+
501
+ def extract_function_calling(rec: dict, config: dict) -> dict | None:
502
+ """tools + query + answers format."""
503
+ if config.get("is_tool_calling"):
504
+ # tool-calling-10k: messages list, assistant_calls list, tools list
505
+ msgs = _parse_maybe_json(rec.get("messages"))
506
+ calls = _parse_maybe_json(rec.get("assistant_calls"))
507
+ tools = _parse_maybe_json(rec.get("tools"))
508
+
509
+ # User mesajini bul
510
+ user_msg = None
511
+ if isinstance(msgs, list):
512
+ for m in msgs:
513
+ if isinstance(m, dict):
514
+ role = m.get("role", "")
515
+ content = m.get("content", "") or m.get("value", "")
516
+ if role in ("user", "human") and content:
517
+ user_msg = content
518
+ break
519
+ if not user_msg:
520
+ return None
521
+ if not calls:
522
+ return None
523
+
524
+ tools_str = _to_json_str(tools) if tools else ""
525
+ calls_str = _to_json_str(calls)
526
+ if not calls_str or len(calls_str) < 10:
527
+ return None
528
+
529
+ sys_msg = (
530
+ "Sen yardımcı bir asistansın. Şu araçları kullanabilirsin:\n\n"
531
+ f"{tools_str}\n\n"
532
+ "Uygun fonksiyonu çağırarak yanıtla."
533
+ ) if tools_str else "Sen yardımcı bir asistansın. Araç çağrılarıyla yanıtla."
534
+
535
+ return {"system": sys_msg, "user": user_msg.strip(), "assistant": calls_str}
536
+
537
+ # function-calling-20k: 3 string field
538
+ tools = rec.get(config.get("tools_key", "tools"), "")
539
+ query = rec.get(config.get("query_key", "query"), "")
540
+ answers = rec.get(config.get("answers_key", "answers"), "")
541
+ if not tools or not query or not answers:
542
+ return None
543
+ if not isinstance(query, str) or len(query.strip()) < 10:
544
+ return None
545
+ sys_msg = (
546
+ "Sen yardımcı bir asistansın. Şu araçları kullanabilirsin:\n\n"
547
+ f"{tools}\n\n"
548
+ "Uygun fonksiyonu çağırarak yanıtla."
549
+ )
550
+ return {
551
+ "system": sys_msg,
552
+ "user": query.strip(),
553
+ "assistant": answers if isinstance(answers, str) else _to_json_str(answers),
554
+ }
555
+
556
+
557
+ def extract_oasst(rec: dict, config: dict) -> dict | None:
558
+ """OpenAssistant tree: parent_id + role + lang + text."""
559
+ if rec.get("lang") != config.get("lang_filter", "tr"):
560
+ return None
561
+ # OAsst tree: ana mesaji al, cevabi al — tek-turn'lik
562
+ # Burada sadece "prompter" tipi root + 1 assistant cevabi alacagiz
563
+ role = rec.get("role", "")
564
+ text = rec.get("text", "")
565
+ if role != "prompter" or not text:
566
+ return None
567
+ # Tek-record extract zor — gercek pipeline tree icin daha karmasik
568
+ # Basitlestirme: prompter mesajini user yap, dummy assistant
569
+ # Bu yontem yetersiz olabilir — ya tum tree'yi paralel oku ya skip
570
+ return None # OAsst kompleks, simdilik skip
571
+
572
+
573
+ def extract_lima(rec: dict, config: dict) -> dict | None:
574
+ """LIMA: conversations [str, str] format."""
575
+ convs = rec.get("conversations") or rec.get("messages")
576
+ if isinstance(convs, list) and len(convs) >= 2:
577
+ if isinstance(convs[0], str):
578
+ return {"user": convs[0].strip(), "assistant": convs[1].strip()}
579
+ return extract_messages_format(rec, config)
580
+ # Fallback standart alanlar
581
+ user = find_field(rec, config.get("input_keys", []))
582
+ asst = find_field(rec, config.get("output_keys", []))
583
+ if user and asst:
584
+ return {"user": user, "assistant": asst}
585
+ return None
586
+
587
+
588
+ def extract_essay(rec: dict, config: dict) -> dict | None:
589
+ """Bilkent / masallar — text alanindan instruction olustur."""
590
+ text = find_field(rec, config.get("text_keys", []) or config.get("output_keys", []))
591
+ if not text:
592
+ # Fallback: ilk uzun string alan
593
+ for k, v in rec.items():
594
+ if isinstance(v, str) and len(v) > 200:
595
+ text = v
596
+ break
597
+ if not text or len(text) < 200:
598
+ return None
599
+ title = find_field(rec, ["title", "name", "baslik"])
600
+
601
+ prompts = []
602
+ if "masallar" in config["repo"].lower():
603
+ prompts = ["Bir masal anlat.", "Bana bir Türk masalı anlat.",
604
+ "Geleneksel bir masal yaz.", "Eski bir masal hikayesi yaz."]
605
+ if title:
606
+ prompts = [f"'{title}' masalını anlat."]
607
+ else:
608
+ prompts = ["Aşağıdaki konuda detaylı bir kompozisyon yaz.",
609
+ "Bu konu hakkında bir yazı yaz."]
610
+ prompt = random.choice(prompts)
611
+ if title and not any(title in p for p in prompts):
612
+ prompt = f"{prompt}\n\nKonu: {title}"
613
+ return {"user": prompt, "assistant": text[:5000]}
614
+
615
+
616
+ def extract_with_thinking(rec: dict, config: dict) -> dict | None:
617
+ """Thinking traces ile: user + thinking + final."""
618
+ user = find_field(rec, config.get("input_keys", []))
619
+ if not user:
620
+ return None
621
+ asst = find_field(rec, config.get("output_keys", []))
622
+ thinking = find_field(rec, [config.get("thinking_key", "thinking")])
623
+ if not asst:
624
+ return None
625
+ # Eger thinking ayri alanda → ChatML icinde <think> tag'i ile birlestir
626
+ if thinking and thinking != asst:
627
+ full_asst = f"<think>\n{thinking}\n</think>\n\n{asst}"
628
+ else:
629
+ full_asst = asst
630
+ return {"user": user, "assistant": full_asst}
631
+
632
+
633
+ def extract_default(rec: dict, config: dict) -> dict | None:
634
+ """Standart input/output extraction."""
635
+ user = find_field(rec, config.get("input_keys", []))
636
+ if not user:
637
+ return None
638
+ asst = find_field(rec, config.get("output_keys", []))
639
+ if not asst:
640
+ return None
641
+
642
+ if "input2_keys" in config:
643
+ extra = find_field(rec, config["input2_keys"])
644
+ if extra and len(extra) > 5 and extra not in user:
645
+ user = f"{user}\n\n{extra}"
646
+
647
+ sys_msg = find_field(rec, config.get("system_keys", []) or ["system"])
648
+
649
+ out = {"user": user, "assistant": asst}
650
+ if sys_msg:
651
+ out["system"] = sys_msg
652
+ return out
653
+
654
+
655
+ def extract_thinking_data(rec: dict, config: dict) -> dict | None:
656
+ """ThinkingData-200K: messages list (system+user) + reasoning + answer."""
657
+ msgs = rec.get("messages", [])
658
+ reasoning = rec.get("reasoning", "")
659
+ answer = rec.get("answer", "")
660
+ if not msgs or not answer:
661
+ return None
662
+ user_msg = ""
663
+ sys_msg = ""
664
+ for m in msgs:
665
+ if not isinstance(m, dict):
666
+ continue
667
+ role = m.get("role", "")
668
+ content = m.get("content", "")
669
+ if role == "user":
670
+ user_msg = content
671
+ elif role == "system":
672
+ sys_msg = content
673
+ if not user_msg:
674
+ return None
675
+ # Combine reasoning + answer with <think> format
676
+ if reasoning and reasoning.strip():
677
+ full_asst = f"<think>\n{reasoning.strip()}\n</think>\n\n{answer.strip()}"
678
+ else:
679
+ full_asst = answer.strip()
680
+ out = {"user": user_msg, "assistant": full_asst}
681
+ if sys_msg:
682
+ out["system"] = sys_msg
683
+ return out
684
+
685
+
686
+ def extract_tot(rec: dict, config: dict) -> dict | None:
687
+ """Tree of Thought: chat_format JSON string."""
688
+ chat_fmt = rec.get("chat_format", "")
689
+ if not chat_fmt:
690
+ # Fallback: question + judged_answer
691
+ q = rec.get("complex_question_tr", "")
692
+ ja = rec.get("judged_answer", {})
693
+ if isinstance(ja, dict):
694
+ ja_text = ja.get("final_answer", "") or ja.get("answer", "") or str(ja)
695
+ else:
696
+ ja_text = str(ja)
697
+ if q and ja_text:
698
+ return {"user": q.strip(), "assistant": ja_text.strip()}
699
+ return None
700
+ # chat_format is a JSON string with user/assistant
701
+ try:
702
+ parsed = json.loads(chat_fmt) if isinstance(chat_fmt, str) else chat_fmt
703
+ if isinstance(parsed, dict):
704
+ # Genelde {messages: [...]} veya {user: ..., assistant: ...}
705
+ if "messages" in parsed:
706
+ return extract_messages_format({"messages": parsed["messages"]}, config)
707
+ if "user" in parsed and "assistant" in parsed:
708
+ return {"user": parsed["user"], "assistant": parsed["assistant"]}
709
+ if isinstance(parsed, list):
710
+ return extract_messages_format({"messages": parsed}, config)
711
+ except Exception:
712
+ pass
713
+ return None
714
+
715
+
716
+ def extract_gpqa(rec: dict) -> dict | None:
717
+ """GPQA: Question + Correct Answer + Incorrect Answer 1-3."""
718
+ question = rec.get("Question", "")
719
+ correct = rec.get("Correct Answer", "")
720
+ incorrects = [
721
+ rec.get("Incorrect Answer 1", ""),
722
+ rec.get("Incorrect Answer 2", ""),
723
+ rec.get("Incorrect Answer 3", ""),
724
+ ]
725
+ incorrects = [x for x in incorrects if x]
726
+ if not question or not correct or len(incorrects) < 2:
727
+ return None
728
+ # Karistir
729
+ choices = [correct] + incorrects
730
+ random.shuffle(choices)
731
+ correct_idx = choices.index(correct)
732
+ correct_letter = chr(65 + correct_idx)
733
+
734
+ choices_text = "\n".join(
735
+ f"{chr(65+i)}) {c}" for i, c in enumerate(choices)
736
+ )
737
+ user = f"{question}\n\n{choices_text}"
738
+ assistant = f"Cevap: {correct_letter}) {correct}"
739
+ return {"user": user, "assistant": assistant}
740
+
741
+
742
+ def extract_quardo(rec: dict, config: dict) -> dict | None:
743
+ """Quardo/Turkish-Chat_GPT-4O: prompt(system) + question(user) + chat(assistant).
744
+ 'chat' alani genelde liste (multi-turn) ya da string olabilir."""
745
+ prompt = rec.get("prompt", "") or ""
746
+ question = rec.get("question", "") or ""
747
+ chat = rec.get("chat", None)
748
+
749
+ # 'chat' bazen JSON string, bazen list of dicts {role, content} olabilir
750
+ asst_text = ""
751
+ if isinstance(chat, str):
752
+ s = chat.strip()
753
+ if s.startswith("[") or s.startswith("{"):
754
+ try:
755
+ chat = json.loads(s)
756
+ except Exception:
757
+ asst_text = s
758
+ else:
759
+ asst_text = s
760
+ if isinstance(chat, list):
761
+ # Ilk assistant mesajini al
762
+ for m in chat:
763
+ if isinstance(m, dict):
764
+ role = m.get("role", "") or m.get("from", "")
765
+ content = m.get("content", "") or m.get("value", "")
766
+ if role in ("assistant", "gpt", "bot") and content:
767
+ asst_text = content
768
+ break
769
+
770
+ if not asst_text:
771
+ # Fallback: 'content' alanini dene
772
+ asst_text = rec.get("content", "") or ""
773
+ if not isinstance(asst_text, str) or len(asst_text.strip()) < 15:
774
+ return None
775
+ if not isinstance(question, str) or len(question.strip()) < 5:
776
+ return None
777
+
778
+ out = {"user": question.strip(), "assistant": asst_text.strip()}
779
+ if prompt and isinstance(prompt, str) and len(prompt.strip()) > 10:
780
+ out["system"] = prompt.strip()
781
+ return out
782
+
783
+
784
+ def extract_inst_format(rec: dict, config: dict) -> dict | None:
785
+ """[INST] ... [/INST] formatindan parse et (Mistral/Llama2 stili)."""
786
+ text = rec.get("text", "")
787
+ if not text or "[INST]" not in text:
788
+ return None
789
+ # <s>[INST] question [/INST] answer </s>
790
+ m = re.search(r"\[INST\]\s*(.+?)\s*\[/INST\]\s*(.+?)\s*(?:</s>|$)",
791
+ text, re.DOTALL)
792
+ if not m:
793
+ return None
794
+ user = m.group(1).strip()
795
+ asst = m.group(2).strip().rstrip("</s>").strip()
796
+ if len(user) < 5 or len(asst) < 5:
797
+ return None
798
+ return {"user": user, "assistant": asst}
799
+
800
+
801
+ def extract_record(rec: dict, config: dict) -> dict | None:
802
+ """Dispatcher — config'e gore dogru handler'i cagir."""
803
+ if config.get("is_mmlu"):
804
+ return extract_mmlu(rec)
805
+ if config.get("is_gpqa"):
806
+ return extract_gpqa(rec)
807
+ if config.get("is_function_calling"):
808
+ return extract_function_calling(rec, config)
809
+ if config.get("is_oasst"):
810
+ return extract_oasst(rec, config)
811
+ if config.get("is_lima"):
812
+ return extract_lima(rec, config)
813
+ if config.get("is_thinking_data"):
814
+ return extract_thinking_data(rec, config)
815
+ if config.get("is_tot"):
816
+ return extract_tot(rec, config)
817
+ if config.get("is_inst_format"):
818
+ return extract_inst_format(rec, config)
819
+ if config.get("is_quardo"):
820
+ return extract_quardo(rec, config)
821
+ if config.get("is_essay"):
822
+ return extract_essay(rec, config)
823
+ if config.get("is_messages"):
824
+ result = extract_messages_format(rec, config)
825
+ if result is not None:
826
+ return result
827
+ return extract_default(rec, config)
828
+ if config.get("thinking_key"):
829
+ return extract_with_thinking(rec, config)
830
+ return extract_default(rec, config)
831
+
832
+
833
+ # =====================================================================
834
+ # Filtreler
835
+ # =====================================================================
836
+ BAD_PATTERNS = [
837
+ r"https?://[^\s]+",
838
+ r"www\.[^\s]+\.[a-z]+",
839
+ r"\b\w+@\w+\.\w+\b",
840
+ r"Erişim tarihi[:.]?\s*\d",
841
+ r"Ana Sayfa\s+(Yaşam|Kadın|Spor)",
842
+ r"Devamını oku|Kaynak:|Reklam",
843
+ r"\b(KDV dahil|Ücretsiz Kargo)\b",
844
+ r"\[\s*reklam\s*\]",
845
+ r"<[^>]{1,50}>",
846
+ ]
847
+
848
+
849
+ def has_repetition(text: str, n: int = 4) -> bool:
850
+ sentences = re.split(r"[.!?]\s+", text)
851
+ if len(sentences) < n:
852
+ return False
853
+ for i in range(len(sentences) - n + 1):
854
+ if len(set(sentences[i:i+n])) == 1 and len(sentences[i]) > 20:
855
+ return True
856
+ return False
857
+
858
+
859
+ def turkish_char_ratio(text: str) -> float:
860
+ if not text:
861
+ return 0.0
862
+ tr_chars = sum(c in "şğıöüçŞĞİÖÜÇ" for c in text)
863
+ letters = sum(c.isalpha() for c in text)
864
+ return tr_chars / max(letters, 1)
865
+
866
+
867
+ def is_turkish(text: str, min_ratio: float = 0.01) -> bool:
868
+ if len(text) < 30:
869
+ return True
870
+ tr_words = {"ve", "bir", "için", "ile", "bu", "şu", "ama", "veya",
871
+ "olan", "olarak", "kadar", "her", "var", "gibi", "yok",
872
+ "ben", "sen", "biz", "siz", "onlar", "ise", "değil",
873
+ "daha", "çok", "az", "diye", "ki", "de", "da"}
874
+ words = re.findall(r"\b\w+\b", text.lower())
875
+ if not words:
876
+ return False
877
+ tr_count = sum(1 for w in words if w in tr_words)
878
+ ratio = tr_count / len(words)
879
+ return ratio >= min_ratio or turkish_char_ratio(text) >= 0.02
880
+
881
+
882
+ def quality_check(rec: dict, is_function_calling: bool = False,
883
+ category: str = "general") -> tuple[bool, str]:
884
+ user = rec.get("user", "")
885
+ asst = rec.get("assistant", "")
886
+
887
+ # Kategoriye gore farkli max uzunluk limitleri
888
+ asst_max = {
889
+ "reasoning": 25000, # thinking traces uzun olabilir
890
+ "curated": 15000, # alican_sft uzun cevaplar
891
+ "knowledge": 10000,
892
+ "conversation": 8000,
893
+ "general": 6000,
894
+ "function": 12000,
895
+ }.get(category, 8000)
896
+
897
+ # Boyut
898
+ if len(user) < 8:
899
+ return False, "user_too_short"
900
+ if len(user) > 4000:
901
+ return False, "user_too_long"
902
+ if len(asst) < 15:
903
+ return False, "asst_too_short"
904
+ if len(asst) > asst_max:
905
+ return False, "asst_too_long"
906
+
907
+ # Function calling: JSON kontrolu
908
+ if is_function_calling:
909
+ s = asst.strip()
910
+ if not (s.startswith("[") or s.startswith("{")):
911
+ return False, "fc_not_json"
912
+ return True, "ok"
913
+
914
+ # Turkce
915
+ if not is_turkish(asst):
916
+ return False, "not_turkish"
917
+
918
+ # Kotu pattern'ler — reasoning/curated icin URL/email check'i atla
919
+ # (akademik citation, kod URLs olabilir; <think> tag normaldir)
920
+ skip_patterns_for = {"reasoning", "curated"}
921
+ if category not in skip_patterns_for:
922
+ for pat in BAD_PATTERNS:
923
+ if re.search(pat, asst):
924
+ return False, "bad_pattern"
925
+ else:
926
+ # Sadece gerçek SEO/spam pattern'leri — HTML/tag check'i YOK
927
+ # <think>, <|begin_of_thought|> gibi tag'ler reasoning icin normaldir
928
+ seo_patterns = [
929
+ r"Ana Sayfa\s+(Yaşam|Kadın|Spor)",
930
+ r"Devamını oku\s*»|Kaynak:\s*http|Reklam\s*\[",
931
+ r"\[\s*reklam\s*\]",
932
+ # Gercek HTML tag'leri (div, p, span, a, br, img, table, tr, td, h1-6)
933
+ r"<(?:div|p|span|a|br|img|table|tr|td|h[1-6])\b[^>]*>",
934
+ ]
935
+ for pat in seo_patterns:
936
+ if re.search(pat, asst):
937
+ return False, "bad_pattern"
938
+
939
+ if has_repetition(asst):
940
+ return False, "repetition"
941
+
942
+ if len(asst) > 200 and turkish_char_ratio(asst) < 0.005:
943
+ return False, "no_turkish_chars"
944
+
945
+ if user.lower() == asst.lower():
946
+ return False, "echo"
947
+
948
+ # Trivial cevap (reasoning + curated hariç)
949
+ if category not in ("reasoning", "curated"):
950
+ if len(asst) < 40 and asst.count(".") <= 1 and asst.count(" ") < 5:
951
+ return False, "asst_trivial"
952
+
953
+ # asst_incomplete: math/code/reasoning icin gevsek kontrol
954
+ # Mid-word kesinti varsa kabul etme — son 30 karakterde bosluk yoksa supheli
955
+ if len(asst) > 50:
956
+ if category in ("reasoning", "conversation", "curated"):
957
+ # Daha gevsek: sadece mid-word kontrol
958
+ last30 = asst[-30:].rstrip()
959
+ # Eger son karakter alfabetik VE bir bosluk yoksa = kesilmis kelime
960
+ if (last30 and last30[-1].isalpha() and " " not in last30
961
+ and last30[-1] not in "ıİĞğüÜşŞçÇöÖ"):
962
+ # Sayilarla bitiyorsa OK (=42, =100)
963
+ pass
964
+ # Cok kisa son satir kesinti
965
+ if len(last30) < 3:
966
+ pass # OK
967
+ # Cogu durumda kabul et
968
+ else:
969
+ # General/knowledge/function: katı kontrol
970
+ if asst[-1] not in ".!?\"')]}…」、。0123456789":
971
+ return False, "asst_incomplete"
972
+
973
+ if asst.count("\n\n\n") > 2:
974
+ return False, "asst_excessive_newlines"
975
+
976
+ short_refusals = ["bilmiyorum", "yardımcı olamam", "anlamadım"]
977
+ if len(asst) < 60 and any(r in asst.lower() for r in short_refusals):
978
+ return False, "asst_refusal_too_short"
979
+
980
+ return True, "ok"
981
+
982
+
983
+ # =====================================================================
984
+ # Pipeline
985
+ # =====================================================================
986
+ def hash_text(text: str) -> str:
987
+ return hashlib.md5(text.lower().strip()[:500].encode()).hexdigest()
988
+
989
+
990
+ def process_dataset(name: str, config: dict, show_progress: bool = True,
991
+ use_cache: bool = True, force: bool = False):
992
+ cache_path = CACHE_DIR / f"{name}.jsonl"
993
+ cache_stats_path = CACHE_DIR / f"{name}_stats.json"
994
+
995
+ # Cache hit — onceden basariyla isleyen dataset'i tekrar indirme
996
+ if use_cache and not force and cache_path.exists():
997
+ n_lines = 0
998
+ accepted = []
999
+ try:
1000
+ with open(cache_path, "r", encoding="utf-8") as f:
1001
+ for line in f:
1002
+ accepted.append(json.loads(line))
1003
+ n_lines += 1
1004
+ print(f"\n{'='*60}\n[{name}] CACHE HIT — {n_lines:,} ornek "
1005
+ f"({cache_path.name})\n{'='*60}")
1006
+ stats = {"cached": True, "accepted": n_lines}
1007
+ if cache_stats_path.exists():
1008
+ with open(cache_stats_path) as sf:
1009
+ stats.update(json.load(sf))
1010
+ return accepted, stats
1011
+ except Exception as e:
1012
+ print(f" ! Cache okunamadi ({e}), yeniden indirilecek")
1013
+
1014
+ print(f"\n{'='*60}\n[{name}] {config['repo']} ({config['category']})\n{'='*60}")
1015
+ try:
1016
+ # config_name varsa subset secimi yap
1017
+ load_kwargs = {"split": config["split"]}
1018
+ if config.get("config_name"):
1019
+ load_kwargs["name"] = config["config_name"]
1020
+ ds = load_dataset(config["repo"], **load_kwargs)
1021
+ except Exception as e:
1022
+ try:
1023
+ # validation veya test split dene
1024
+ for alt_split in ["validation", "test", "train"]:
1025
+ if alt_split == config["split"]:
1026
+ continue
1027
+ try:
1028
+ load_kwargs["split"] = alt_split
1029
+ ds = load_dataset(config["repo"], **load_kwargs)
1030
+ print(f" (split degisti: {config['split']} -> {alt_split})")
1031
+ break
1032
+ except Exception:
1033
+ continue
1034
+ else:
1035
+ raise e
1036
+ except Exception as e2:
1037
+ print(f" ! Yuklenemedi: {e2}")
1038
+ return [], {"error": str(e2)}
1039
+
1040
+ total = len(ds)
1041
+ target = config.get("target", total)
1042
+ print(f" Toplam: {total:,} satir | hedef: {target:,}")
1043
+
1044
+ accepted = []
1045
+ reject_reasons = Counter()
1046
+ rejected_samples = []
1047
+ seen_hashes = set()
1048
+ is_fc = config.get("is_function_calling", False)
1049
+ cat = config["category"]
1050
+
1051
+ iterator = ds if not show_progress else tqdm(ds, total=total, desc=f" {name}")
1052
+ for rec in iterator:
1053
+ # Erken durma — target'a ulaştık
1054
+ if len(accepted) >= target * 1.5: # %50 fazla topla, sonra sample
1055
+ break
1056
+
1057
+ extracted = extract_record(rec, config)
1058
+ if extracted is None:
1059
+ reject_reasons["no_fields"] += 1
1060
+ continue
1061
+
1062
+ # Source-spesifik ek filtreler (config'ten gelir)
1063
+ if config.get("skip_translation"):
1064
+ uq = extracted.get("user", "")
1065
+ if any(kw in uq for kw in ("çevirin", "çevir ", "çeviriniz",
1066
+ "tercüme", "Translate", "translate",
1067
+ "limerik")):
1068
+ reject_reasons["translation_filtered"] += 1
1069
+ continue
1070
+ min_alen = config.get("min_asst_len")
1071
+ if min_alen and len(extracted.get("assistant", "")) < min_alen:
1072
+ reject_reasons["below_min_asst_len"] += 1
1073
+ continue
1074
+
1075
+ ok, reason = quality_check(extracted, is_function_calling=is_fc, category=cat)
1076
+ if not ok:
1077
+ reject_reasons[reason] += 1
1078
+ if len(rejected_samples) < 3:
1079
+ rejected_samples.append({
1080
+ "reason": reason,
1081
+ "user": extracted.get("user", "")[:200],
1082
+ "assistant": extracted.get("assistant", "")[:200],
1083
+ })
1084
+ continue
1085
+
1086
+ h = hash_text(extracted["user"] + extracted["assistant"][:200])
1087
+ if h in seen_hashes:
1088
+ reject_reasons["duplicate"] += 1
1089
+ continue
1090
+ seen_hashes.add(h)
1091
+
1092
+ extracted["source"] = name
1093
+ extracted["category"] = cat
1094
+ if config.get("tag"):
1095
+ extracted["tag"] = config["tag"]
1096
+ accepted.append(extracted)
1097
+
1098
+ # Target'a indir (random sample)
1099
+ if len(accepted) > target:
1100
+ random.seed(42)
1101
+ accepted = random.sample(accepted, target)
1102
+
1103
+ stats = {
1104
+ "repo": config["repo"],
1105
+ "category": cat,
1106
+ "total_loaded": total,
1107
+ "accepted": len(accepted),
1108
+ "accept_rate": len(accepted) / max(total, 1),
1109
+ "target": target,
1110
+ "rejects": dict(reject_reasons.most_common(10)),
1111
+ "rejected_samples": rejected_samples,
1112
+ }
1113
+ print(f" Accept: {len(accepted):,} (target {target}) | "
1114
+ f"raw {sum(reject_reasons.values())} reject")
1115
+ top3 = reject_reasons.most_common(3)
1116
+ print(f" Top rejects: {top3}")
1117
+
1118
+ # Cache'e yaz — sadece basarili (>0 ornek) sonuclari
1119
+ if len(accepted) > 0:
1120
+ try:
1121
+ with open(cache_path, "w", encoding="utf-8") as f:
1122
+ for r in accepted:
1123
+ f.write(json.dumps(r, ensure_ascii=False) + "\n")
1124
+ with open(cache_stats_path, "w", encoding="utf-8") as sf:
1125
+ json.dump(stats, sf, indent=2, ensure_ascii=False)
1126
+ print(f" [cache] {cache_path.name} yazildi")
1127
+ except Exception as e:
1128
+ print(f" ! Cache yazma hatasi: {e}")
1129
+ return accepted, stats
1130
+
1131
+
1132
+ def to_chatml(rec: dict) -> dict:
1133
+ msgs = []
1134
+ if rec.get("system"):
1135
+ msgs.append({"role": "system", "content": rec["system"]})
1136
+ msgs.append({"role": "user", "content": rec["user"]})
1137
+ msgs.append({"role": "assistant", "content": rec["assistant"]})
1138
+ return {
1139
+ "messages": msgs,
1140
+ "source": rec.get("source", "unknown"),
1141
+ "category": rec.get("category", "general"),
1142
+ "tag": rec.get("tag", ""),
1143
+ }
1144
+
1145
+
1146
+ def main():
1147
+ parser = argparse.ArgumentParser()
1148
+ parser.add_argument("--datasets", nargs="+", default=None,
1149
+ help="Sadece secili datasets (varsayilan: hepsi)")
1150
+ parser.add_argument("--categories", nargs="+", default=None,
1151
+ choices=CATEGORIES,
1152
+ help="Sadece secili kategoriler")
1153
+ parser.add_argument("--target", type=int, default=150000,
1154
+ help="Hedef TOPLAM ornek sayisi (varsayilan: 150K)")
1155
+ parser.add_argument("--out", type=str,
1156
+ default=str(DATA_DIR / "01_collected.jsonl"))
1157
+ parser.add_argument("--no-progress", action="store_true")
1158
+ parser.add_argument("--skip-failed", action="store_true",
1159
+ help="Yuklenemeyen datasets'i sessizce atla")
1160
+ parser.add_argument("--force", nargs="*", default=None,
1161
+ help="Bu dataset'leri cache'e ragmen yeniden indir "
1162
+ "(boş list → tumu, isim list → sadece bunlar)")
1163
+ parser.add_argument("--clear-cache", action="store_true",
1164
+ help="Cache'i tamamen sil ve sifirdan basla")
1165
+ parser.add_argument("--no-cache", action="store_true",
1166
+ help="Cache kullanma, her seferinde yeniden indir")
1167
+ args = parser.parse_args()
1168
+
1169
+ if args.clear_cache:
1170
+ import shutil
1171
+ if CACHE_DIR.exists():
1172
+ shutil.rmtree(CACHE_DIR)
1173
+ CACHE_DIR.mkdir(parents=True, exist_ok=True)
1174
+ print(f"Cache temizlendi: {CACHE_DIR}")
1175
+
1176
+ random.seed(42)
1177
+
1178
+ # Dataset secimi
1179
+ if args.datasets:
1180
+ selected = {n: c for n, c in DATASETS.items() if n in args.datasets}
1181
+ elif args.categories:
1182
+ selected = {n: c for n, c in DATASETS.items()
1183
+ if c["category"] in args.categories}
1184
+ else:
1185
+ selected = DATASETS
1186
+
1187
+ print(f"Toplanacak datasets: {len(selected)}")
1188
+ print(f"Kategoriler: {Counter(c['category'] for c in selected.values())}")
1189
+
1190
+ all_records = []
1191
+ all_stats = {}
1192
+
1193
+ # Force list — boş list → hepsini, name list → sadece bunları
1194
+ force_set = None
1195
+ if args.force is not None:
1196
+ force_set = set(args.force) if args.force else set(selected.keys())
1197
+
1198
+ for name, config in selected.items():
1199
+ force_this = force_set is not None and name in force_set
1200
+ records, stats = process_dataset(
1201
+ name, config,
1202
+ show_progress=not args.no_progress,
1203
+ use_cache=not args.no_cache,
1204
+ force=force_this,
1205
+ )
1206
+ all_records.extend(records)
1207
+ all_stats[name] = stats
1208
+ if "error" in stats and not args.skip_failed:
1209
+ print(f" ! {name} yuklenemedi — devam edilecek")
1210
+
1211
+ # Cross-dataset dedup
1212
+ print(f"\n{'='*60}\nCROSS-DATASET DEDUP\n{'='*60}")
1213
+ seen = set()
1214
+ deduped = []
1215
+ for rec in all_records:
1216
+ h = hash_text(rec["user"] + rec["assistant"][:200])
1217
+ if h not in seen:
1218
+ seen.add(h)
1219
+ deduped.append(rec)
1220
+ removed = len(all_records) - len(deduped)
1221
+ print(f" Cross-dataset dedup: {removed:,} silindi")
1222
+ all_records = deduped
1223
+
1224
+ # Kategori bazli ozet (final shuffle once)
1225
+ cat_counts_before = Counter(r["category"] for r in all_records)
1226
+ print(f"\nKategori bazli (filter sonrasi):")
1227
+ for cat, n in cat_counts_before.most_common():
1228
+ print(f" {cat:<15} {n:>7,}")
1229
+
1230
+ # Target'a uydur (stratified by category)
1231
+ if args.target and len(all_records) > args.target:
1232
+ # Target kategori oranlarini koru
1233
+ # mmlu gated, oasst skip → ayarlı oranlar:
1234
+ # general(40) + reasoning(50) + knowledge(15) + conversation(20) + curated(10) + function(5) ≈ 140
1235
+ cat_target = {
1236
+ "general": int(args.target * 40 / 140),
1237
+ "reasoning": int(args.target * 50 / 140),
1238
+ "knowledge": int(args.target * 15 / 140),
1239
+ "conversation": int(args.target * 20 / 140),
1240
+ "curated": int(args.target * 10 / 140),
1241
+ "function": int(args.target * 5 / 140),
1242
+ }
1243
+ # Eger args.target != 150K, oranlar 150K bazli, normalizasyon
1244
+ by_cat = defaultdict(list)
1245
+ for r in all_records:
1246
+ by_cat[r["category"]].append(r)
1247
+ sampled = []
1248
+ for cat, recs in by_cat.items():
1249
+ random.shuffle(recs)
1250
+ n = min(cat_target.get(cat, len(recs)), len(recs))
1251
+ sampled.extend(recs[:n])
1252
+ all_records = sampled
1253
+ print(f"\n Stratified downsample: {len(all_records):,}")
1254
+
1255
+ random.shuffle(all_records)
1256
+
1257
+ # Yaz
1258
+ out_path = Path(args.out)
1259
+ out_path.parent.mkdir(parents=True, exist_ok=True)
1260
+ with open(out_path, "w", encoding="utf-8") as f:
1261
+ for rec in all_records:
1262
+ f.write(json.dumps(to_chatml(rec), ensure_ascii=False) + "\n")
1263
+ print(f"\n[OK] Yazildi: {out_path}")
1264
+ print(f" {len(all_records):,} ornek")
1265
+
1266
+ # Source + category dagilim
1267
+ src_counts = Counter(r["source"] for r in all_records)
1268
+ cat_counts = Counter(r["category"] for r in all_records)
1269
+ print(f"\n Kategori dagilim:")
1270
+ for cat, n in cat_counts.most_common():
1271
+ print(f" {cat:<15} {n:>7,}")
1272
+ print(f"\n Source dagilim:")
1273
+ for src, n in src_counts.most_common(30):
1274
+ print(f" {src:<25} {n:>7,}")
1275
+
1276
+ # Stats kaydet
1277
+ stats_summary = {
1278
+ "total_collected": len(all_records),
1279
+ "by_category": dict(cat_counts),
1280
+ "by_source": dict(src_counts),
1281
+ "datasets": all_stats,
1282
+ }
1283
+ stats_path = out_path.with_name(out_path.stem.replace("collected", "stats") + ".json")
1284
+ with open(stats_path, "w", encoding="utf-8") as f:
1285
+ json.dump(stats_summary, f, indent=2, ensure_ascii=False)
1286
+ print(f" Stats: {stats_path}")
1287
+
1288
+ # Rejected ornekler
1289
+ rejected_path = out_path.with_name(out_path.stem.replace("collected", "rejected_samples") + ".jsonl")
1290
+ with open(rejected_path, "w", encoding="utf-8") as f:
1291
+ for name, stats in all_stats.items():
1292
+ for sample in stats.get("rejected_samples", []):
1293
+ sample["dataset"] = name
1294
+ f.write(json.dumps(sample, ensure_ascii=False) + "\n")
1295
+ print(f" Rejected sample inceleme: {rejected_path}")
1296
+
1297
+ print(f"\nSonraki adim:")
1298
+ print(f" python sft_02_finalize.py (kalite kontrol + ChatML push)")
1299
+
1300
+
1301
+ if __name__ == "__main__":
1302
+ main()