File size: 813 Bytes
b62c338
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
from best_of_n_generator import generate_n
from best_selector import select_best
import json

DATASET = []

PROMPTS = [
    "LRU 캐시 구현",
    "다익스트라 설명",
    "FastAPI 서버 설계",
    "Redis 구조"
]

for epoch in range(3):

    new_data = []

    for p in PROMPTS:

        # 1. N개 생성
        samples = generate_n(p, n=5)

        # 2. best 선택
        best = select_best(samples)

        new_data.append({
            "instruction": p,
            "output": best
        })

    # 3. 데이터 누적
    DATASET += new_data

    print(f"Epoch {epoch} complete:", len(DATASET))

# 저장
with open("dataset_final.jsonl", "w", encoding="utf-8") as f:
    for d in DATASET:
        f.write(json.dumps(d, ensure_ascii=False) + "\n")

print("[DONE] self-improving dataset ready")