File size: 4,704 Bytes
7ac6a19
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
#!/usr/bin/env python3
"""
Server-side Hugging Face Datacenter Ingestion in Robust Batches:
Specialized Physics, Chemistry, and Biology Pretraining Stream -> science/
1. NCERT Class 6-12 Physics, Chemistry, Biology in Hindi & English (oss-codes/NCERT-Conversational-Dataset-Indic)
2. Pure Science Textbooks 30 Parquet Shards (RedMod/science_textbooks - 11.85 GB)
3. SciQ Scientific QA & Explanations (allenai/sciq)
"""

import time
from huggingface_hub import HfApi, CommitOperationCopy

DEST_REPO = "ViuAI/viu-mini-raw-pretrain"
REPO_TYPE = "dataset"
BATCH_SIZE = 6  # Small chunk size to avoid remote HTTP connection reset (WinError 10054)

def commit_batch(api, operations, batch_idx, total_batches, commit_prefix):
    max_retries = 4
    for attempt in range(1, max_retries + 1):
        try:
            print(f"[*] Committing batch {batch_idx}/{total_batches} ({len(operations)} operations, attempt {attempt})...")
            commit_info = api.create_commit(
                repo_id=DEST_REPO,
                repo_type=REPO_TYPE,
                operations=operations,
                commit_message=f"{commit_prefix} (Batch {batch_idx}/{total_batches})",
            )
            print(f"[OK] Batch {batch_idx}/{total_batches} committed: {commit_info.commit_url}")
            time.sleep(2)
            return True
        except Exception as e:
            print(f"[!] Batch {batch_idx} attempt {attempt} failed: {e}")
            if attempt < max_retries:
                time.sleep(5 * attempt)
            else:
                raise e

def main():
    api = HfApi()
    print(f"[*] Preparing operations for destination repo: {DEST_REPO}")
    operations = []

    # 1. NCERT Science Files (Hindi & English)
    ncert_repo = "oss-codes/NCERT-Conversational-Dataset-Indic"
    ncert_files = api.list_repo_files(ncert_repo, repo_type="dataset")

    science_keywords = [
        "physics", "chemistry", "biology", "science - vi", "science - vii", 
        "science - viii", "science - ix", "science-x"
    ]

    selected_ncert = []
    for f in ncert_files:
        if f.startswith("hi/") or f.startswith("en/"):
            f_lower = f.lower()
            if any(k in f_lower for k in science_keywords):
                if "political" in f_lower or "social" in f_lower or "human ecology" in f_lower:
                    continue
                selected_ncert.append(f)

    print(f"[*] Selected {len(selected_ncert)} NCERT Science files (Hindi & English).")
    for f in selected_ncert:
        lang = "hindi" if f.startswith("hi/") else "english"
        base_name = f.split("/")[-1].replace("_sharegpt_conversations.jsonl", ".jsonl").replace(" ", "_").lower()
        dest_path = f"science/ncert_{lang}_{base_name}"
        operations.append(
            CommitOperationCopy(
                src_repo_id=ncert_repo,
                src_path_in_repo=f,
                path_in_repo=dest_path,
                src_repo_type="dataset",
            )
        )

    # 2. SciQ Question Answering & Reasoning (Physics, Chemistry, Biology)
    sciq_repo = "allenai/sciq"
    sciq_files = [
        ("data/train-00000-of-00001.parquet", "science/sciq_train.parquet"),
        ("data/validation-00000-of-00001.parquet", "science/sciq_validation.parquet"),
        ("data/test-00000-of-00001.parquet", "science/sciq_test.parquet"),
    ]
    for src_file, dest_file in sciq_files:
        operations.append(
            CommitOperationCopy(
                src_repo_id=sciq_repo,
                src_path_in_repo=src_file,
                path_in_repo=dest_file,
                src_repo_type="dataset",
            )
        )

    # 3. RedMod Pure Science Textbooks (30 Parquet Shards, 11.85 GB)
    redmod_repo = "RedMod/science_textbooks"
    for i in range(30):
        src_file = f"part-{i:05d}.parquet"
        dest_file = f"science/textbooks_part_{i:05d}.parquet"
        operations.append(
            CommitOperationCopy(
                src_repo_id=redmod_repo,
                src_path_in_repo=src_file,
                path_in_repo=dest_file,
                src_repo_type="dataset",
            )
        )

    print(f"[*] Total operations to execute: {len(operations)}")
    batches = [operations[i:i + BATCH_SIZE] for i in range(0, len(operations), BATCH_SIZE)]
    print(f"[*] Split into {len(batches)} batches of up to {BATCH_SIZE} operations each.")

    t0 = time.time()
    for b_idx, b_ops in enumerate(batches, start=1):
        commit_batch(api, b_ops, b_idx, len(batches), "Add science stream datasets")

    elapsed = time.time() - t0
    print(f"[OK] All {len(batches)} batches successfully committed in {elapsed:.2f} seconds!")

if __name__ == "__main__":
    main()