BHARGAV REDDY commited on
Commit
b361816
·
verified ·
1 Parent(s): d134029

Upload Base/Datasets/rag_mcp_sft/build_rag_mcp_sft_dataset.py with huggingface_hub

Browse files
Base/Datasets/rag_mcp_sft/build_rag_mcp_sft_dataset.py ADDED
@@ -0,0 +1,1123 @@
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
+ import argparse
2
+ import hashlib
3
+ import json
4
+ import random
5
+ import re
6
+ from pathlib import Path
7
+
8
+ from transformers import AutoTokenizer
9
+
10
+
11
+ ROOT = Path(__file__).resolve().parents[3]
12
+ DATASET_DIR = Path(__file__).resolve().parent
13
+ TOKENIZER_DIR = ROOT / "Base" / "checkpoints" / "EleutherAI" / "pythia-160m"
14
+ RETRIEVED_ON = "2026-04-03"
15
+
16
+
17
+ SOURCE_MANIFEST = [
18
+ {
19
+ "topic": "mcp",
20
+ "url": "https://modelcontextprotocol.io/introduction",
21
+ "title": "What is the Model Context Protocol (MCP)?",
22
+ "notes": [
23
+ "MCP is an open standard for connecting AI applications to tools, data sources, and workflows.",
24
+ "The ecosystem spans hosts like IDEs and assistants plus many local and remote servers.",
25
+ "The main value is interoperable context and actions across tools without bespoke integrations.",
26
+ ],
27
+ },
28
+ {
29
+ "topic": "mcp",
30
+ "url": "https://modelcontextprotocol.io/docs/concepts/architecture",
31
+ "title": "MCP Architecture Overview",
32
+ "notes": [
33
+ "MCP uses a client-server architecture with hosts, clients, and servers.",
34
+ "The data layer is JSON-RPC 2.0 based and includes lifecycle, tools, resources, prompts, and notifications.",
35
+ "The transport layer covers stdio for local use and streamable HTTP for remote use.",
36
+ ],
37
+ },
38
+ {
39
+ "topic": "rag",
40
+ "url": "https://www.pinecone.io/learn/retrieval-augmented-generation/",
41
+ "title": "Retrieval-Augmented Generation (RAG)",
42
+ "notes": [
43
+ "RAG addresses stale knowledge, weak domain depth, missing private data, and hallucination risk.",
44
+ "Core stages are ingestion, retrieval, augmentation, and generation.",
45
+ "Good RAG systems emphasize chunking, retrieval quality, reranking, grounding, citations, and evaluation.",
46
+ ],
47
+ },
48
+ ]
49
+
50
+
51
+ KNOWLEDGE_CARDS = [
52
+ {
53
+ "id": "rag_definition",
54
+ "topic": "RAG",
55
+ "title": "Definition of RAG",
56
+ "summary": "Retrieval-augmented generation improves model answers by retrieving external evidence before response generation.",
57
+ "key_points": [
58
+ "RAG grounds the model with authoritative context instead of relying only on frozen weights.",
59
+ "It is useful when answers need current, proprietary, or domain-specific information.",
60
+ "The goal is higher accuracy, relevance, and trustworthiness.",
61
+ ],
62
+ "pitfalls": [
63
+ "Using weak sources still produces weak answers.",
64
+ "Large context dumps can increase cost without improving quality.",
65
+ "A model can still hallucinate if retrieval is poor or ignored.",
66
+ ],
67
+ "best_practices": [
68
+ "Prefer authoritative sources with clear provenance.",
69
+ "Ask the model to stay grounded in retrieved facts.",
70
+ "Return citations or source identifiers when possible.",
71
+ ],
72
+ "use_cases": [
73
+ "support copilots",
74
+ "enterprise knowledge assistants",
75
+ "research workflows",
76
+ ],
77
+ },
78
+ {
79
+ "id": "rag_limitations",
80
+ "topic": "RAG",
81
+ "title": "Why base models need retrieval",
82
+ "summary": "Standalone language models are limited by knowledge cutoffs, incomplete domain depth, and lack of private data access.",
83
+ "key_points": [
84
+ "A pretrained model cannot know events or documents added after training.",
85
+ "Specialized domains often require evidence that was sparse or absent in pretraining.",
86
+ "Private company knowledge should not be expected to appear in a public base model.",
87
+ ],
88
+ "pitfalls": [
89
+ "Users may trust fluent but unsupported answers.",
90
+ "Sampling randomness can surface plausible but incorrect claims.",
91
+ "Without citations, error diagnosis is harder.",
92
+ ],
93
+ "best_practices": [
94
+ "Add retrieval when freshness or policy accuracy matters.",
95
+ "Use evaluations with known-good answers.",
96
+ "Treat fluent output as unverified until grounded.",
97
+ ],
98
+ "use_cases": [
99
+ "policy assistants",
100
+ "product documentation bots",
101
+ "current-events question answering",
102
+ ],
103
+ },
104
+ {
105
+ "id": "rag_ingestion",
106
+ "topic": "RAG",
107
+ "title": "Ingestion stage",
108
+ "summary": "The ingestion stage prepares authoritative documents so retrieval can find them efficiently later.",
109
+ "key_points": [
110
+ "Ingestion usually includes cleaning, chunking, embedding, and indexing.",
111
+ "Source metadata such as title, owner, and timestamp should be preserved.",
112
+ "Fresh content should trigger incremental updates rather than full rebuilds when possible.",
113
+ ],
114
+ "pitfalls": [
115
+ "Dropping metadata weakens traceability.",
116
+ "Poor chunking reduces retrieval quality.",
117
+ "Stale indexes silently degrade answers.",
118
+ ],
119
+ "best_practices": [
120
+ "Keep a documented ingestion contract.",
121
+ "Track source versioning and refresh policy.",
122
+ "Store enough metadata for filtering and citations.",
123
+ ],
124
+ "use_cases": [
125
+ "document onboarding pipelines",
126
+ "knowledge base refresh jobs",
127
+ "multi-tenant content indexing",
128
+ ],
129
+ },
130
+ {
131
+ "id": "rag_chunking",
132
+ "topic": "RAG",
133
+ "title": "Chunking strategy",
134
+ "summary": "Chunking balances retrieval precision against context completeness by dividing documents into useful units.",
135
+ "key_points": [
136
+ "Small chunks improve focus but may lose context.",
137
+ "Large chunks preserve context but can dilute retrieval precision.",
138
+ "Overlap can preserve continuity across boundaries.",
139
+ ],
140
+ "pitfalls": [
141
+ "Fixed-size chunks can split critical facts mid-thought.",
142
+ "Too much overlap increases storage and duplicate retrieval.",
143
+ "Ignoring document structure wastes natural section boundaries.",
144
+ ],
145
+ "best_practices": [
146
+ "Align chunking with document structure when possible.",
147
+ "Test chunk size against real user queries.",
148
+ "Avoid chunk sizes that exceed downstream prompt budgets.",
149
+ ],
150
+ "use_cases": [
151
+ "API docs",
152
+ "wiki articles",
153
+ "policy manuals",
154
+ ],
155
+ },
156
+ {
157
+ "id": "rag_embeddings",
158
+ "topic": "RAG",
159
+ "title": "Embeddings and vector search",
160
+ "summary": "Embeddings turn text into vectors so semantically related passages can be retrieved even when wording differs.",
161
+ "key_points": [
162
+ "Dense embeddings capture semantic similarity.",
163
+ "The embedding model should match the language and domain of the corpus.",
164
+ "Vector search is often paired with metadata filtering.",
165
+ ],
166
+ "pitfalls": [
167
+ "Bad embeddings cause systematic retrieval misses.",
168
+ "Ignoring metadata can surface irrelevant tenants or time periods.",
169
+ "Switching embedding models without reindexing corrupts similarity assumptions.",
170
+ ],
171
+ "best_practices": [
172
+ "Version embedding models and indexes together.",
173
+ "Benchmark retrieval before production rollout.",
174
+ "Use metadata filters for scope control.",
175
+ ],
176
+ "use_cases": [
177
+ "semantic search",
178
+ "similar policy lookup",
179
+ "knowledge retrieval for assistants",
180
+ ],
181
+ },
182
+ {
183
+ "id": "rag_hybrid_retrieval",
184
+ "topic": "RAG",
185
+ "title": "Hybrid retrieval and reranking",
186
+ "summary": "Hybrid retrieval combines semantic and lexical signals, then reranking sharpens the final evidence set.",
187
+ "key_points": [
188
+ "Dense retrieval helps when users paraphrase concepts.",
189
+ "Lexical or sparse retrieval helps with exact codes, product names, and acronyms.",
190
+ "Reranking can improve final relevance after broad recall.",
191
+ ],
192
+ "pitfalls": [
193
+ "Dense-only search may miss literal identifiers.",
194
+ "Keyword-only search may miss semantically related wording.",
195
+ "Skipping rerank can leave the top context noisy.",
196
+ ],
197
+ "best_practices": [
198
+ "Treat retrieval as a pipeline, not a single score.",
199
+ "Log both recall and final answer quality.",
200
+ "Tune retrieval for the actual query distribution.",
201
+ ],
202
+ "use_cases": [
203
+ "incident knowledge search",
204
+ "developer documentation search",
205
+ "internal ticket assistants",
206
+ ],
207
+ },
208
+ {
209
+ "id": "rag_augmentation",
210
+ "topic": "RAG",
211
+ "title": "Augmented prompting",
212
+ "summary": "Augmentation combines the user question with retrieved context in a prompt that tells the model how to use evidence.",
213
+ "key_points": [
214
+ "The prompt should distinguish the question from the evidence block.",
215
+ "The model should be instructed to say when context is insufficient.",
216
+ "Answer style, citation format, and refusal behavior should be explicit.",
217
+ ],
218
+ "pitfalls": [
219
+ "Prompt stuffing wastes tokens and distracts the model.",
220
+ "Unclear grounding rules encourage unsupported guesses.",
221
+ "Mixing contradictory passages without guidance confuses answer synthesis.",
222
+ ],
223
+ "best_practices": [
224
+ "Use a consistent prompt contract.",
225
+ "Cap context to the most relevant evidence.",
226
+ "Require the model to state uncertainty when evidence is missing.",
227
+ ],
228
+ "use_cases": [
229
+ "grounded question answering",
230
+ "support responses with references",
231
+ "document-based summarization",
232
+ ],
233
+ },
234
+ {
235
+ "id": "rag_generation",
236
+ "topic": "RAG",
237
+ "title": "Generation and grounded answers",
238
+ "summary": "The generation stage turns the augmented prompt into an answer that should remain faithful to retrieved evidence.",
239
+ "key_points": [
240
+ "Generation should prioritize grounded facts over fluent speculation.",
241
+ "Well-designed RAG answers can include citations or source mentions.",
242
+ "The system should avoid pretending the evidence said more than it did.",
243
+ ],
244
+ "pitfalls": [
245
+ "Ungrounded synthesis reintroduces hallucinations.",
246
+ "Verbose answers can hide unsupported claims.",
247
+ "Dropping citation signals weakens user trust.",
248
+ ],
249
+ "best_practices": [
250
+ "Keep answers scoped to retrieved evidence.",
251
+ "Surface missing information plainly.",
252
+ "Design output formats that make verification easy.",
253
+ ],
254
+ "use_cases": [
255
+ "answer generation with citations",
256
+ "enterprise assistant responses",
257
+ "grounded summaries",
258
+ ],
259
+ },
260
+ {
261
+ "id": "rag_evaluation",
262
+ "topic": "RAG",
263
+ "title": "RAG evaluation",
264
+ "summary": "RAG quality should be measured with representative queries, known-good answers, and retrieval diagnostics.",
265
+ "key_points": [
266
+ "Evaluation should cover both retrieval quality and answer quality.",
267
+ "A fixed validation set helps track regressions over time.",
268
+ "Ground truth is essential for knowing whether improvements are real.",
269
+ ],
270
+ "pitfalls": [
271
+ "Only tracking latency or token cost misses quality failures.",
272
+ "Synthetic benchmarks alone may not reflect production behavior.",
273
+ "Ignoring bad-query cases hides operational risk.",
274
+ ],
275
+ "best_practices": [
276
+ "Maintain a curated benchmark set.",
277
+ "Review failures by retrieval stage and answer stage.",
278
+ "Track citation accuracy and insufficiency handling.",
279
+ ],
280
+ "use_cases": [
281
+ "release validation",
282
+ "retrieval tuning",
283
+ "regression testing",
284
+ ],
285
+ },
286
+ {
287
+ "id": "rag_agentic",
288
+ "topic": "RAG",
289
+ "title": "Agentic RAG",
290
+ "summary": "Agentic RAG extends basic retrieval by letting an agent plan, query, validate, and refine evidence over multiple steps.",
291
+ "key_points": [
292
+ "An agent can rewrite queries, choose tools, and validate retrieved context.",
293
+ "Iterative retrieval helps when a single search step is not enough.",
294
+ "The system can combine retrieval with reasoning and execution tools.",
295
+ ],
296
+ "pitfalls": [
297
+ "More steps can increase cost and latency.",
298
+ "Poor control loops can amplify mistakes instead of correcting them.",
299
+ "Without guardrails, tool use can become noisy or unsafe.",
300
+ ],
301
+ "best_practices": [
302
+ "Use agentic loops only when the task benefits from iteration.",
303
+ "Record tool traces for debugging.",
304
+ "Stop early when evidence is already sufficient.",
305
+ ],
306
+ "use_cases": [
307
+ "deep research agents",
308
+ "multi-step support diagnosis",
309
+ "workflow automation with retrieval",
310
+ ],
311
+ },
312
+ {
313
+ "id": "mcp_definition",
314
+ "topic": "MCP",
315
+ "title": "Definition of MCP",
316
+ "summary": "The Model Context Protocol is an open standard that lets AI applications connect to external tools, resources, and workflows.",
317
+ "key_points": [
318
+ "MCP standardizes integration between AI applications and external systems.",
319
+ "It supports tools, data resources, and reusable prompts.",
320
+ "The design goal is interoperability instead of custom one-off connectors.",
321
+ ],
322
+ "pitfalls": [
323
+ "MCP is not a model architecture or a retrieval algorithm.",
324
+ "Using the protocol does not remove the need for access control.",
325
+ "A bad tool design remains bad even with a standard protocol.",
326
+ ],
327
+ "best_practices": [
328
+ "Use MCP when you need structured tool or context integration.",
329
+ "Keep server capabilities explicit and documented.",
330
+ "Design tools for clarity and predictable behavior.",
331
+ ],
332
+ "use_cases": [
333
+ "IDE assistants",
334
+ "enterprise chat agents",
335
+ "local tool orchestration",
336
+ ],
337
+ },
338
+ {
339
+ "id": "mcp_value",
340
+ "topic": "MCP",
341
+ "title": "Why MCP matters",
342
+ "summary": "MCP reduces integration cost by giving clients and servers a shared protocol for context exchange and actions.",
343
+ "key_points": [
344
+ "Developers can build once and integrate across multiple hosts.",
345
+ "AI applications gain structured access to data and actions.",
346
+ "End users benefit from more capable assistants with fewer brittle integrations.",
347
+ ],
348
+ "pitfalls": [
349
+ "Protocol support alone does not guarantee a good user experience.",
350
+ "Overexposing server capabilities can increase risk.",
351
+ "Lack of schema clarity makes tool use unreliable.",
352
+ ],
353
+ "best_practices": [
354
+ "Publish precise capability schemas.",
355
+ "Design for portability across hosts.",
356
+ "Treat auth and permissions as first-class concerns.",
357
+ ],
358
+ "use_cases": [
359
+ "cross-platform agent integrations",
360
+ "editor tooling",
361
+ "internal workflow assistants",
362
+ ],
363
+ },
364
+ {
365
+ "id": "mcp_architecture",
366
+ "topic": "MCP",
367
+ "title": "Hosts, clients, and servers",
368
+ "summary": "MCP follows a client-server model where a host application manages one client connection per server.",
369
+ "key_points": [
370
+ "The host is the AI application coordinating the experience.",
371
+ "The client manages the protocol connection to a server.",
372
+ "The server exposes context or actions to the client.",
373
+ ],
374
+ "pitfalls": [
375
+ "Confusing the host with the server leads to incorrect designs.",
376
+ "Treating all servers as interchangeable hides capability differences.",
377
+ "Ignoring connection lifecycle creates brittle integrations.",
378
+ ],
379
+ "best_practices": [
380
+ "Map host, client, and server roles explicitly in architecture docs.",
381
+ "Store capability metadata per connection.",
382
+ "Separate protocol management from model reasoning logic.",
383
+ ],
384
+ "use_cases": [
385
+ "VS Code plus filesystem server",
386
+ "assistant plus database server",
387
+ "multi-server agent hosts",
388
+ ],
389
+ },
390
+ {
391
+ "id": "mcp_layers",
392
+ "topic": "MCP",
393
+ "title": "Data layer and transport layer",
394
+ "summary": "MCP separates the JSON-RPC data protocol from the transport that carries those messages.",
395
+ "key_points": [
396
+ "The data layer defines requests, responses, notifications, and primitives.",
397
+ "The transport layer handles framing, connection establishment, and authentication.",
398
+ "The same protocol semantics can work over different transports.",
399
+ ],
400
+ "pitfalls": [
401
+ "Mixing protocol semantics with transport specifics reduces portability.",
402
+ "Transport security must not be assumed from protocol semantics alone.",
403
+ "State handling still matters even when transport is abstracted.",
404
+ ],
405
+ "best_practices": [
406
+ "Keep the wire protocol independent from business logic.",
407
+ "Document which capabilities depend on transport features.",
408
+ "Test both local and remote connection paths when supported.",
409
+ ],
410
+ "use_cases": [
411
+ "local stdio tools",
412
+ "remote HTTP services",
413
+ "protocol portability",
414
+ ],
415
+ },
416
+ {
417
+ "id": "mcp_transports",
418
+ "topic": "MCP",
419
+ "title": "Stdio and streamable HTTP",
420
+ "summary": "MCP commonly uses stdio for local integrations and streamable HTTP for remote services.",
421
+ "key_points": [
422
+ "Stdio is efficient for local process-to-process communication.",
423
+ "Streamable HTTP fits remote servers and standard web auth patterns.",
424
+ "OAuth or bearer-style auth is relevant for remote deployments.",
425
+ ],
426
+ "pitfalls": [
427
+ "Choosing remote transport for a purely local tool adds avoidable overhead.",
428
+ "Ignoring auth on remote transports is a major security error.",
429
+ "Different deployment models need different observability plans.",
430
+ ],
431
+ "best_practices": [
432
+ "Use stdio for local single-machine workflows.",
433
+ "Use HTTP when multi-client remote access is required.",
434
+ "Align transport choice with operational boundaries.",
435
+ ],
436
+ "use_cases": [
437
+ "local filesystem servers",
438
+ "remote SaaS integrations",
439
+ "hybrid agent stacks",
440
+ ],
441
+ },
442
+ {
443
+ "id": "mcp_lifecycle",
444
+ "topic": "MCP",
445
+ "title": "Lifecycle and initialization",
446
+ "summary": "MCP is stateful and begins with an initialization exchange that negotiates protocol version and capabilities.",
447
+ "key_points": [
448
+ "Initialization establishes compatibility between client and server.",
449
+ "Capability negotiation tells each side which primitives are supported.",
450
+ "Identity fields help with debugging and version tracking.",
451
+ ],
452
+ "pitfalls": [
453
+ "Skipping initialization breaks capability assumptions.",
454
+ "Protocol version mismatches can cause subtle failures.",
455
+ "Stateful connections require explicit teardown logic.",
456
+ ],
457
+ "best_practices": [
458
+ "Fail fast on incompatible versions.",
459
+ "Persist negotiated capabilities in connection state.",
460
+ "Emit readiness only after initialization succeeds.",
461
+ ],
462
+ "use_cases": [
463
+ "server registration",
464
+ "capability discovery",
465
+ "session bootstrap",
466
+ ],
467
+ },
468
+ {
469
+ "id": "mcp_server_primitives",
470
+ "topic": "MCP",
471
+ "title": "Server primitives",
472
+ "summary": "Servers can expose tools, resources, and prompts so hosts can discover and use them systematically.",
473
+ "key_points": [
474
+ "Tools represent executable actions.",
475
+ "Resources represent contextual data sources.",
476
+ "Prompts represent reusable interaction templates.",
477
+ ],
478
+ "pitfalls": [
479
+ "Blurry tool descriptions reduce model reliability.",
480
+ "Resources without scope control can leak sensitive data.",
481
+ "Prompts should not hide important operational assumptions.",
482
+ ],
483
+ "best_practices": [
484
+ "Name tools clearly and uniquely.",
485
+ "Provide explicit schemas and descriptions.",
486
+ "Expose only the primitives that serve real workflows.",
487
+ ],
488
+ "use_cases": [
489
+ "calculator tools",
490
+ "schema resources",
491
+ "prompt libraries",
492
+ ],
493
+ },
494
+ {
495
+ "id": "mcp_client_primitives",
496
+ "topic": "MCP",
497
+ "title": "Client primitives",
498
+ "summary": "Clients can support sampling, elicitation, and logging so servers can participate in richer workflows.",
499
+ "key_points": [
500
+ "Sampling lets a server request model completions from the host side.",
501
+ "Elicitation lets a server ask the user for additional information.",
502
+ "Logging lets servers send debugging or monitoring information.",
503
+ ],
504
+ "pitfalls": [
505
+ "Server authors should not assume every client supports every primitive.",
506
+ "Unbounded elicitation can frustrate users.",
507
+ "Logs can leak sensitive values if not scrubbed.",
508
+ ],
509
+ "best_practices": [
510
+ "Check negotiated capabilities before using client-side features.",
511
+ "Use elicitation only when the missing information matters.",
512
+ "Keep logs structured and privacy-aware.",
513
+ ],
514
+ "use_cases": [
515
+ "model-independent server logic",
516
+ "interactive approval flows",
517
+ "debuggable integrations",
518
+ ],
519
+ },
520
+ {
521
+ "id": "mcp_tool_discovery",
522
+ "topic": "MCP",
523
+ "title": "Tool discovery",
524
+ "summary": "Clients discover tools through list operations before attempting execution.",
525
+ "key_points": [
526
+ "Discovery avoids guessing what a server supports.",
527
+ "Tool metadata should include name, title, description, and input schema.",
528
+ "Hosts can build a unified tool registry from multiple servers.",
529
+ ],
530
+ "pitfalls": [
531
+ "Calling undeclared tools is brittle.",
532
+ "Weak descriptions make tool selection harder for the model.",
533
+ "Ignoring schema validation increases runtime failures.",
534
+ ],
535
+ "best_practices": [
536
+ "Refresh discovery when capabilities change.",
537
+ "Use descriptive tool names.",
538
+ "Validate tool arguments against the published schema.",
539
+ ],
540
+ "use_cases": [
541
+ "tool registries",
542
+ "dynamic agent capabilities",
543
+ "multi-server hosts",
544
+ ],
545
+ },
546
+ {
547
+ "id": "mcp_tool_execution",
548
+ "topic": "MCP",
549
+ "title": "Tool execution",
550
+ "summary": "After discovery, a client can call a tool with structured arguments and receive structured content back.",
551
+ "key_points": [
552
+ "Tool calls reference the exact discovered tool name.",
553
+ "Arguments should conform to the tool's JSON schema.",
554
+ "Responses can contain structured content, not just plain text.",
555
+ ],
556
+ "pitfalls": [
557
+ "Ambiguous tool names make routing error-prone.",
558
+ "Malformed arguments create avoidable failures.",
559
+ "Treating tool output as trusted without validation can be risky.",
560
+ ],
561
+ "best_practices": [
562
+ "Keep tool contracts stable.",
563
+ "Return actionable, well-typed results.",
564
+ "Capture execution traces for debugging and safety review.",
565
+ ],
566
+ "use_cases": [
567
+ "weather lookup",
568
+ "database queries",
569
+ "filesystem operations",
570
+ ],
571
+ },
572
+ {
573
+ "id": "mcp_notifications",
574
+ "topic": "MCP",
575
+ "title": "Notifications and live updates",
576
+ "summary": "Notifications let servers tell clients about changes, such as tool list updates, without waiting for a request.",
577
+ "key_points": [
578
+ "Notifications are one-way JSON-RPC messages with no response expected.",
579
+ "They are useful when available capabilities change dynamically.",
580
+ "Clients typically refresh their local registry after a change notification.",
581
+ ],
582
+ "pitfalls": [
583
+ "Clients that ignore notifications can become stale.",
584
+ "Sending change events without rate control can create noise.",
585
+ "Capabilities should declare whether list-changed notifications are supported.",
586
+ ],
587
+ "best_practices": [
588
+ "Use notifications for meaningful state changes.",
589
+ "Tie notification handling to cache refresh logic.",
590
+ "Keep event semantics explicit.",
591
+ ],
592
+ "use_cases": [
593
+ "dynamic tool catalogs",
594
+ "live server state updates",
595
+ "responsive agent interfaces",
596
+ ],
597
+ },
598
+ {
599
+ "id": "mcp_security",
600
+ "topic": "MCP",
601
+ "title": "Security and governance",
602
+ "summary": "MCP integrations still need authentication, authorization, scope control, and careful data handling.",
603
+ "key_points": [
604
+ "Remote servers should use appropriate authentication mechanisms.",
605
+ "Servers should expose only the minimum necessary capabilities.",
606
+ "Operational logs and tool arguments can contain sensitive data.",
607
+ ],
608
+ "pitfalls": [
609
+ "Assuming protocol standardization removes security work is incorrect.",
610
+ "Overbroad tools can create unnecessary blast radius.",
611
+ "Weak auditability slows incident response.",
612
+ ],
613
+ "best_practices": [
614
+ "Apply least privilege.",
615
+ "Audit tool usage and sensitive accesses.",
616
+ "Separate local trusted tools from remote untrusted surfaces.",
617
+ ],
618
+ "use_cases": [
619
+ "enterprise governance",
620
+ "remote connector security",
621
+ "access-controlled assistants",
622
+ ],
623
+ },
624
+ {
625
+ "id": "bridge_mcp_rag",
626
+ "topic": "Bridge",
627
+ "title": "MCP plus RAG",
628
+ "summary": "MCP and RAG solve different layers of the stack: RAG grounds answers with retrieved evidence, while MCP standardizes how tools and context services are connected.",
629
+ "key_points": [
630
+ "RAG is a retrieval-and-generation pattern.",
631
+ "MCP is an interoperability protocol.",
632
+ "A retrieval service can be exposed through MCP as a tool or resource.",
633
+ ],
634
+ "pitfalls": [
635
+ "MCP does not replace retrieval design.",
636
+ "RAG does not replace integration contracts.",
637
+ "Confusing protocol choice with answer quality leads to bad planning.",
638
+ ],
639
+ "best_practices": [
640
+ "Use RAG for grounding and MCP for integration boundaries.",
641
+ "Keep retrieval quality metrics separate from protocol compatibility metrics.",
642
+ "Expose retrieval services through clear MCP schemas when integrating agents.",
643
+ ],
644
+ "use_cases": [
645
+ "agentic research assistants",
646
+ "tool-based knowledge systems",
647
+ "retrieval-backed IDE helpers",
648
+ ],
649
+ },
650
+ {
651
+ "id": "bridge_retrieval_server",
652
+ "topic": "Bridge",
653
+ "title": "Retrieval as an MCP server",
654
+ "summary": "A retrieval engine can be packaged as an MCP server so hosts can discover search tools and evidence resources dynamically.",
655
+ "key_points": [
656
+ "An MCP server can expose search tools with structured query schemas.",
657
+ "Retrieved passages can be returned as typed content with metadata.",
658
+ "This keeps retrieval integration reusable across hosts.",
659
+ ],
660
+ "pitfalls": [
661
+ "If the retrieval server lacks filters, tenants may leak across queries.",
662
+ "Search results without provenance weaken trust.",
663
+ "Large unranked result sets inflate token cost.",
664
+ ],
665
+ "best_practices": [
666
+ "Return ranked snippets with metadata.",
667
+ "Expose filters explicitly in the tool schema.",
668
+ "Keep retrieval latency visible to the host.",
669
+ ],
670
+ "use_cases": [
671
+ "shared search backends",
672
+ "multi-host retrieval reuse",
673
+ "retrieval-backed agent tools",
674
+ ],
675
+ },
676
+ {
677
+ "id": "bridge_agent_design",
678
+ "topic": "Bridge",
679
+ "title": "Agent design with RAG and MCP",
680
+ "summary": "A strong agent architecture often uses RAG for evidence and MCP for structured tool access around that evidence.",
681
+ "key_points": [
682
+ "The agent can retrieve evidence, call tools, and synthesize a grounded answer.",
683
+ "MCP helps the host manage multiple external capabilities consistently.",
684
+ "RAG helps the model reason with fresher and narrower evidence.",
685
+ ],
686
+ "pitfalls": [
687
+ "Too many tools without routing logic can overwhelm the system.",
688
+ "Too much retrieved context can bury the key answer.",
689
+ "Without evaluation, complex agent loops hide regressions.",
690
+ ],
691
+ "best_practices": [
692
+ "Start with a simple grounded path before adding loop complexity.",
693
+ "Measure both tool success and answer faithfulness.",
694
+ "Keep the final answer tied to cited evidence or tool outputs.",
695
+ ],
696
+ "use_cases": [
697
+ "engineering copilots",
698
+ "support agents",
699
+ "analyst assistants",
700
+ ],
701
+ },
702
+ ]
703
+
704
+
705
+ AUDIENCES = [
706
+ "a junior ML engineer",
707
+ "a backend engineer",
708
+ "a product manager",
709
+ "an enterprise architect",
710
+ "a solutions engineer",
711
+ "an AI platform lead",
712
+ ]
713
+
714
+ CONSTRAINTS = [
715
+ "Keep the answer practical and free of hype.",
716
+ "Use plain English with no jargon overload.",
717
+ "Make the answer useful for implementation planning.",
718
+ "Focus on operational tradeoffs, not marketing language.",
719
+ "Explain the idea cleanly without drifting into unrelated topics.",
720
+ ]
721
+
722
+ FOCUSES = [
723
+ "definition and purpose",
724
+ "system design choices",
725
+ "deployment tradeoffs",
726
+ "quality and evaluation",
727
+ "security and governance",
728
+ "how to explain it to a team",
729
+ ]
730
+
731
+ QUESTION_OPENERS = [
732
+ "What is",
733
+ "Explain",
734
+ "Give a practical explanation of",
735
+ "Teach me",
736
+ "Describe",
737
+ "Help me understand",
738
+ ]
739
+
740
+ DESCRIPTION_OPENERS = [
741
+ "Write a description of",
742
+ "Draft an internal note about",
743
+ "Create a clean explainer for",
744
+ "Produce a concise briefing on",
745
+ "Write a technical overview of",
746
+ ]
747
+
748
+ COMPARISON_OPENERS = [
749
+ "Compare",
750
+ "Explain the difference between",
751
+ "When should a team choose",
752
+ "Contrast",
753
+ ]
754
+
755
+ SCENARIO_OPENERS = [
756
+ "How should a team use",
757
+ "What does good implementation of",
758
+ "How would you apply",
759
+ "Design a practical approach for",
760
+ ]
761
+
762
+ MISCONCEPTION_OPENERS = [
763
+ "Correct this misunderstanding about",
764
+ "What do people often get wrong about",
765
+ "Clarify a common mistake about",
766
+ ]
767
+
768
+
769
+ def normalize_text(text):
770
+ text = text.replace("\r\n", "\n")
771
+ text = re.sub(r"[ \t]+", " ", text)
772
+ text = re.sub(r"\n{3,}", "\n\n", text)
773
+ return text.strip()
774
+
775
+
776
+ def clean_sentence(text):
777
+ return text.strip().rstrip(".")
778
+
779
+
780
+ def format_entry(entry):
781
+ parts = [f"### Instruction:\n{entry['instruction'].strip()}"]
782
+ if entry.get("input"):
783
+ parts.append(f"### Input:\n{entry['input'].strip()}")
784
+ parts.append(f"### Response:\n{entry['output'].strip()}")
785
+ return "\n\n".join(parts)
786
+
787
+
788
+ def hash_entry(entry):
789
+ key = "|||".join(
790
+ normalize_text(entry.get(field, "")).lower()
791
+ for field in ("instruction", "input", "output")
792
+ )
793
+ return hashlib.sha1(key.encode("utf-8")).hexdigest()
794
+
795
+
796
+ def sentence_join(items):
797
+ return " ".join(item.rstrip(".") + "." for item in items if item)
798
+
799
+
800
+ def choose_subset(values, rng, minimum=2, maximum=3):
801
+ count = min(len(values), rng.randint(minimum, maximum))
802
+ return rng.sample(values, count)
803
+
804
+
805
+ def build_context(rng, extra_focus=None):
806
+ lines = [
807
+ f"Audience: {rng.choice(AUDIENCES)}",
808
+ f"Focus: {extra_focus or rng.choice(FOCUSES)}",
809
+ f"Style rule: {rng.choice(CONSTRAINTS)}",
810
+ ]
811
+ if rng.random() < 0.55:
812
+ lines.append(f"Length target: {rng.choice(['120-180 words', '180-260 words', '220-320 words', '4-6 bullet points'])}")
813
+ return "\n".join(lines)
814
+
815
+
816
+ def build_direct_output(card, rng):
817
+ facts = choose_subset(card["key_points"], rng, 2, 3)
818
+ practices = choose_subset(card["best_practices"], rng, 1, 2)
819
+ opener = f"The concept '{card['title']}' can be understood like this:"
820
+ body = sentence_join([card["summary"], *facts])
821
+ close = "Good practice is to " + ", then ".join(
822
+ clean_sentence(practice[0].lower() + practice[1:] if practice and practice[0].isupper() else practice)
823
+ for practice in practices
824
+ ).rstrip(".") + "."
825
+ return normalize_text(f"{opener} {body} {close}")
826
+
827
+
828
+ def build_description_output(card, rng):
829
+ facts = choose_subset(card["key_points"], rng, 3, 3)
830
+ pitfalls = choose_subset(card["pitfalls"], rng, 1, 2)
831
+ use_cases = choose_subset(card["use_cases"], rng, 2, 3)
832
+ lines = [
833
+ f"The topic '{card['title']}' matters because {card['summary'][0].lower() + card['summary'][1:]}",
834
+ "Key points:",
835
+ ]
836
+ lines.extend(f"- {fact}" for fact in facts)
837
+ lines.append("Why teams care:")
838
+ lines.extend(f"- It supports {use_case}." for use_case in use_cases)
839
+ lines.append("What to avoid:")
840
+ lines.extend(f"- {pitfall}" for pitfall in pitfalls)
841
+ return normalize_text("\n".join(lines))
842
+
843
+
844
+ def build_checklist_output(card, rng):
845
+ practices = choose_subset(card["best_practices"], rng, 3, 3)
846
+ pitfalls = choose_subset(card["pitfalls"], rng, 2, 2)
847
+ lines = [
848
+ f"Use this checklist when working with '{card['title']}':",
849
+ *[f"- {clean_sentence(practice)}." for practice in practices],
850
+ "Watch-outs:",
851
+ *[f"- {clean_sentence(pitfall)}." for pitfall in pitfalls],
852
+ ]
853
+ return normalize_text("\n".join(lines))
854
+
855
+
856
+ def build_scenario_output(card, rng):
857
+ facts = choose_subset(card["key_points"], rng, 2, 3)
858
+ practices = choose_subset(card["best_practices"], rng, 2, 2)
859
+ scenario = rng.choice(card["use_cases"])
860
+ lines = [
861
+ f"For a team building around {scenario}, start by treating '{card['title']}' as an engineering system rather than a vague AI feature.",
862
+ sentence_join([card["summary"], *facts]),
863
+ "A practical rollout would:",
864
+ *[f"- {clean_sentence(practice)}." for practice in practices],
865
+ ]
866
+ return normalize_text("\n".join(lines))
867
+
868
+
869
+ def build_misconception_output(card, rng):
870
+ pitfall = rng.choice(card["pitfalls"])
871
+ facts = choose_subset(card["key_points"], rng, 2, 2)
872
+ return normalize_text(
873
+ f"A common mistake is this: {pitfall} The correction is straightforward. "
874
+ f"{card['summary']} {sentence_join(facts)}"
875
+ )
876
+
877
+
878
+ def build_comparison_output(left, right, rng):
879
+ left_fact = rng.choice(left["key_points"])
880
+ right_fact = rng.choice(right["key_points"])
881
+ left_practice = rng.choice(left["best_practices"])
882
+ right_practice = rng.choice(right["best_practices"])
883
+ lines = [
884
+ f"'{left['title']}' and '{right['title']}' solve different problems, even when they appear in the same AI stack.",
885
+ f"For '{left['title']}': {left['summary']} {left_fact}",
886
+ f"For '{right['title']}': {right['summary']} {right_fact}",
887
+ "Use them together only when the product needs both grounded knowledge and structured external capabilities.",
888
+ f"A sensible rule is to {clean_sentence(left_practice[0].lower() + left_practice[1:] if left_practice and left_practice[0].isupper() else left_practice)} and to {clean_sentence(right_practice[0].lower() + right_practice[1:] if right_practice and right_practice[0].isupper() else right_practice)}.",
889
+ ]
890
+ return normalize_text(" ".join(lines))
891
+
892
+
893
+ def make_question_entry(card, rng):
894
+ instruction = f"{rng.choice(QUESTION_OPENERS)} '{card['title']}' in practical terms."
895
+ return {
896
+ "instruction": instruction,
897
+ "input": build_context(rng),
898
+ "output": build_direct_output(card, rng),
899
+ "kind": "qna",
900
+ "topic": card["topic"],
901
+ "sources": [card["id"]],
902
+ }
903
+
904
+
905
+ def make_description_entry(card, rng):
906
+ instruction = f"{rng.choice(DESCRIPTION_OPENERS)} '{card['title']}' for an engineering handbook."
907
+ return {
908
+ "instruction": instruction,
909
+ "input": build_context(rng, "definition and purpose"),
910
+ "output": build_description_output(card, rng),
911
+ "kind": "description",
912
+ "topic": card["topic"],
913
+ "sources": [card["id"]],
914
+ }
915
+
916
+
917
+ def make_checklist_entry(card, rng):
918
+ instruction = f"Create an implementation checklist for '{card['title']}'."
919
+ return {
920
+ "instruction": instruction,
921
+ "input": build_context(rng, "system design choices"),
922
+ "output": build_checklist_output(card, rng),
923
+ "kind": "checklist",
924
+ "topic": card["topic"],
925
+ "sources": [card["id"]],
926
+ }
927
+
928
+
929
+ def make_scenario_entry(card, rng):
930
+ opener = rng.choice(SCENARIO_OPENERS)
931
+ if opener == "What does good implementation of":
932
+ instruction = f"What does good implementation of '{card['title']}' look like for a real product team?"
933
+ else:
934
+ instruction = f"{opener} '{card['title']}' for a real product team?"
935
+ return {
936
+ "instruction": instruction,
937
+ "input": build_context(rng, "deployment tradeoffs"),
938
+ "output": build_scenario_output(card, rng),
939
+ "kind": "scenario",
940
+ "topic": card["topic"],
941
+ "sources": [card["id"]],
942
+ }
943
+
944
+
945
+ def make_misconception_entry(card, rng):
946
+ instruction = f"{rng.choice(MISCONCEPTION_OPENERS)} '{card['title']}'."
947
+ return {
948
+ "instruction": instruction,
949
+ "input": build_context(rng, "how to explain it to a team"),
950
+ "output": build_misconception_output(card, rng),
951
+ "kind": "clarification",
952
+ "topic": card["topic"],
953
+ "sources": [card["id"]],
954
+ }
955
+
956
+
957
+ def make_comparison_entry(cards, rng):
958
+ left, right = rng.sample(cards, 2)
959
+ instruction = f"{rng.choice(COMPARISON_OPENERS)} '{left['title']}' and '{right['title']}'."
960
+ return {
961
+ "instruction": instruction,
962
+ "input": build_context(rng, "system design choices"),
963
+ "output": build_comparison_output(left, right, rng),
964
+ "kind": "comparison",
965
+ "topic": f"{left['topic']}+{right['topic']}",
966
+ "sources": [left["id"], right["id"]],
967
+ }
968
+
969
+
970
+ ENTRY_BUILDERS = [
971
+ make_question_entry,
972
+ make_description_entry,
973
+ make_checklist_entry,
974
+ make_scenario_entry,
975
+ make_misconception_entry,
976
+ ]
977
+
978
+
979
+ def trim_to_window(entry, tokenizer, max_seq_len):
980
+ output = entry["output"]
981
+ while True:
982
+ total_tokens = len(tokenizer.encode(format_entry(entry)))
983
+ if total_tokens <= max_seq_len:
984
+ return entry, total_tokens
985
+ pieces = re.split(r"(?<=[.!?])\s+", output)
986
+ if len(pieces) <= 2:
987
+ return None, None
988
+ output = " ".join(pieces[:-1]).strip()
989
+ entry["output"] = output
990
+
991
+
992
+ def build_dataset(target_tokens, seed, max_seq_len, val_ratio):
993
+ rng = random.Random(seed)
994
+ tokenizer = AutoTokenizer.from_pretrained(str(TOKENIZER_DIR))
995
+ manifest_path = DATASET_DIR / "source_manifest.json"
996
+ manifest_path.write_text(json.dumps({
997
+ "retrieved_on": RETRIEVED_ON,
998
+ "sources": SOURCE_MANIFEST,
999
+ }, indent=2, ensure_ascii=False), encoding="utf-8")
1000
+
1001
+ accepted = []
1002
+ seen = set()
1003
+ total_tokens = 0
1004
+ attempts = 0
1005
+ builder_weights = [5, 4, 3, 3, 2, 3]
1006
+
1007
+ while total_tokens < target_tokens:
1008
+ attempts += 1
1009
+ builder_index = rng.choices(range(6), weights=builder_weights, k=1)[0]
1010
+ if builder_index == 5:
1011
+ candidate = make_comparison_entry(KNOWLEDGE_CARDS, rng)
1012
+ else:
1013
+ card = rng.choice(KNOWLEDGE_CARDS)
1014
+ candidate = ENTRY_BUILDERS[builder_index](card, rng)
1015
+
1016
+ candidate["instruction"] = normalize_text(candidate["instruction"])
1017
+ candidate["input"] = normalize_text(candidate["input"])
1018
+ candidate["output"] = normalize_text(candidate["output"])
1019
+
1020
+ trimmed, token_count = trim_to_window(candidate, tokenizer, max_seq_len)
1021
+ if trimmed is None or token_count is None or token_count < 80:
1022
+ continue
1023
+
1024
+ dedupe_key = hash_entry(trimmed)
1025
+ if dedupe_key in seen:
1026
+ continue
1027
+
1028
+ seen.add(dedupe_key)
1029
+ accepted.append({
1030
+ "instruction": trimmed["instruction"],
1031
+ "input": trimmed["input"],
1032
+ "output": trimmed["output"],
1033
+ "meta": {
1034
+ "kind": trimmed["kind"],
1035
+ "topic": trimmed["topic"],
1036
+ "sources": trimmed["sources"],
1037
+ "token_count": token_count,
1038
+ },
1039
+ })
1040
+ total_tokens += token_count
1041
+
1042
+ if len(accepted) % 1000 == 0:
1043
+ print(f"accepted={len(accepted):,} tokens={total_tokens:,} attempts={attempts:,}")
1044
+
1045
+ rng.shuffle(accepted)
1046
+ val_size = max(1, int(len(accepted) * val_ratio))
1047
+ val_data = accepted[:val_size]
1048
+ train_data = accepted[val_size:]
1049
+ return train_data, val_data, total_tokens
1050
+
1051
+
1052
+ def write_outputs(train_data, val_data, total_tokens, target_tokens):
1053
+ train_path = DATASET_DIR / "train.json"
1054
+ val_path = DATASET_DIR / "val.json"
1055
+ full_path = DATASET_DIR / "all.json"
1056
+ report_path = DATASET_DIR / "BUILD_REPORT.md"
1057
+ preview_path = DATASET_DIR / "sample_preview.json"
1058
+
1059
+ compact_train = [{k: entry[k] for k in ("instruction", "input", "output")} for entry in train_data]
1060
+ compact_val = [{k: entry[k] for k in ("instruction", "input", "output")} for entry in val_data]
1061
+ compact_full = compact_train + compact_val
1062
+
1063
+ train_path.write_text(json.dumps(compact_train, ensure_ascii=False), encoding="utf-8")
1064
+ val_path.write_text(json.dumps(compact_val, ensure_ascii=False), encoding="utf-8")
1065
+ full_path.write_text(json.dumps(compact_full, ensure_ascii=False), encoding="utf-8")
1066
+ preview_path.write_text(json.dumps((train_data + val_data)[:20], indent=2, ensure_ascii=False), encoding="utf-8")
1067
+
1068
+ avg_tokens = total_tokens / max(1, len(compact_full))
1069
+ by_kind = {}
1070
+ by_topic = {}
1071
+ for entry in train_data + val_data:
1072
+ kind = entry["meta"]["kind"]
1073
+ topic = entry["meta"]["topic"]
1074
+ by_kind[kind] = by_kind.get(kind, 0) + 1
1075
+ by_topic[topic] = by_topic.get(topic, 0) + 1
1076
+
1077
+ lines = [
1078
+ "# RAG + MCP SFT Build Report",
1079
+ "",
1080
+ f"- Retrieved on: {RETRIEVED_ON}",
1081
+ f"- Target tokens: {target_tokens:,}",
1082
+ f"- Realized tokens: {total_tokens:,}",
1083
+ f"- Train samples: {len(train_data):,}",
1084
+ f"- Val samples: {len(val_data):,}",
1085
+ f"- Total samples: {len(compact_full):,}",
1086
+ f"- Average formatted tokens per sample: {avg_tokens:.1f}",
1087
+ f"- Max window enforced: 1024 tokens",
1088
+ "",
1089
+ "## Breakdown by kind",
1090
+ "",
1091
+ ]
1092
+ for kind, count in sorted(by_kind.items()):
1093
+ lines.append(f"- {kind}: {count:,}")
1094
+ lines.extend(["", "## Breakdown by topic", ""])
1095
+ for topic, count in sorted(by_topic.items()):
1096
+ lines.append(f"- {topic}: {count:,}")
1097
+ lines.extend(["", "## Files", "", "- train.json", "- val.json", "- all.json", "- sample_preview.json", "- source_manifest.json"])
1098
+ report_path.write_text("\n".join(lines), encoding="utf-8")
1099
+
1100
+
1101
+ def main():
1102
+ parser = argparse.ArgumentParser(description="Build a clean RAG+MCP SFT dataset for LUNA")
1103
+ parser.add_argument("--target-tokens", type=int, default=10_000_000)
1104
+ parser.add_argument("--seed", type=int, default=42)
1105
+ parser.add_argument("--max-seq-len", type=int, default=1024)
1106
+ parser.add_argument("--val-ratio", type=float, default=0.02)
1107
+ args = parser.parse_args()
1108
+
1109
+ train_data, val_data, total_tokens = build_dataset(
1110
+ target_tokens=args.target_tokens,
1111
+ seed=args.seed,
1112
+ max_seq_len=args.max_seq_len,
1113
+ val_ratio=args.val_ratio,
1114
+ )
1115
+ write_outputs(train_data, val_data, total_tokens, args.target_tokens)
1116
+ print(
1117
+ f"dataset_ready train={len(train_data):,} val={len(val_data):,} total_tokens={total_tokens:,} "
1118
+ f"dir={DATASET_DIR}"
1119
+ )
1120
+
1121
+
1122
+ if __name__ == "__main__":
1123
+ main()