| from best_of_n_generator import generate_n | |
| from best_selector import select_best | |
| import json | |
| DATASET = [] | |
| PROMPTS = [ | |
| "LRU ์บ์ ๊ตฌํ", | |
| "๋ค์ต์คํธ๋ผ ์ค๋ช ", | |
| "FastAPI ์๋ฒ ์ค๊ณ", | |
| "Redis ๊ตฌ์กฐ" | |
| ] | |
| for epoch in range(3): | |
| new_data = [] | |
| for p in PROMPTS: | |
| # 1. N๊ฐ ์์ฑ | |
| samples = generate_n(p, n=5) | |
| # 2. best ์ ํ | |
| best = select_best(samples) | |
| new_data.append({ | |
| "instruction": p, | |
| "output": best | |
| }) | |
| # 3. ๋ฐ์ดํฐ ๋์ | |
| DATASET += new_data | |
| print(f"Epoch {epoch} complete:", len(DATASET)) | |
| # ์ ์ฅ | |
| with open("dataset_final.jsonl", "w", encoding="utf-8") as f: | |
| for d in DATASET: | |
| f.write(json.dumps(d, ensure_ascii=False) + "\n") | |
| print("[DONE] self-improving dataset ready") | |