File size: 2,465 Bytes
c1b8df7
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
import os
import pandas as pd
import datasets


def main():
    OUTPUT_DIR = "masteries/coding/data/raw"
    os.makedirs(OUTPUT_DIR, exist_ok=True)
    OUTPUT_PATH = os.path.join(OUTPUT_DIR, "codecontests_fused_dataset.parquet")

    print("Loading ByteDance-Seed/Code-Contests-Plus dataset (streaming mode)...")
    ds = datasets.load_dataset(
        "ByteDance-Seed/Code-Contests-Plus", split="train", streaming=True
    )

    clean_data = []
    buggy_data = []

    # Target dataset size: 50k clean, 50k buggy
    TARGET_PER_CLASS = 50000

    print(f"Extracting up to {TARGET_PER_CLASS} Python samples per class...")

    for item in ds:
        # Extract correct submissions (Clean / Label 0)
        for sub in item.get("correct_submissions", []):
            if (
                "python" in str(sub.get("language", "")).lower()
                and len(clean_data) < TARGET_PER_CLASS
            ):
                clean_data.append(
                    {"mutated_code": sub.get("code", ""), "label": "CLEAN"}
                )

        # Extract incorrect submissions (Buggy / Label 1)
        for sub in item.get("incorrect_submissions", []):
            if (
                "python" in str(sub.get("language", "")).lower()
                and len(buggy_data) < TARGET_PER_CLASS
            ):
                buggy_data.append({"mutated_code": sub.get("code", ""), "label": "BUG"})

        # Stop early if we have enough data
        if len(clean_data) >= TARGET_PER_CLASS and len(buggy_data) >= TARGET_PER_CLASS:
            break

    print(f"Extracted {len(clean_data)} CLEAN samples.")
    print(f"Extracted {len(buggy_data)} BUGGY samples.")

    df_clean = pd.DataFrame(clean_data)
    df_bugs = pd.DataFrame(buggy_data)

    print("Fusing and randomizing dataset...")
    df_fused = pd.concat([df_clean, df_bugs], ignore_index=True)

    # Remove completely duplicated snippets
    df_fused = df_fused.drop_duplicates(
        subset=["mutated_code"], keep="first"
    ).reset_index(drop=True)

    # Final deep shuffle
    df_fused = df_fused.sample(frac=1, random_state=42).reset_index(drop=True)

    print(f"Saving to {OUTPUT_PATH}...")
    df_fused.to_parquet(OUTPUT_PATH)

    print(f"SUCCESS! Dataset saved to {OUTPUT_PATH}")
    print(f"Total Rows: {len(df_fused)}")
    print(
        f"Clean Data: {(df_fused['label'] == 'CLEAN').sum()} | Bug Data: {(df_fused['label'] == 'BUG').sum()}"
    )


if __name__ == "__main__":
    main()