Download data/scripts/add_science_stem_datasets.py from ViuAI/ViuMini-Dense-360M: direct link, hf CLI and curl.
- Browser
- Download file 4.7 kB
-
https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/add_science_stem_datasets.py
- Command line
-
hf download hf://ViuAI/ViuMini-Dense-360M/data/scripts/add_science_stem_datasets.py
-
curl -L -o add_science_stem_datasets.py https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/add_science_stem_datasets.py
4.7 kB
| #!/usr/bin/env python3 | |
| """ | |
| Server-side Hugging Face Datacenter Ingestion in Robust Batches: | |
| Specialized Physics, Chemistry, and Biology Pretraining Stream -> science/ | |
| 1. NCERT Class 6-12 Physics, Chemistry, Biology in Hindi & English (oss-codes/NCERT-Conversational-Dataset-Indic) | |
| 2. Pure Science Textbooks 30 Parquet Shards (RedMod/science_textbooks - 11.85 GB) | |
| 3. SciQ Scientific QA & Explanations (allenai/sciq) | |
| """ | |
| import time | |
| from huggingface_hub import HfApi, CommitOperationCopy | |
| DEST_REPO = "ViuAI/viu-mini-raw-pretrain" | |
| REPO_TYPE = "dataset" | |
| BATCH_SIZE = 6 # Small chunk size to avoid remote HTTP connection reset (WinError 10054) | |
| def commit_batch(api, operations, batch_idx, total_batches, commit_prefix): | |
| max_retries = 4 | |
| for attempt in range(1, max_retries + 1): | |
| try: | |
| print(f"[*] Committing batch {batch_idx}/{total_batches} ({len(operations)} operations, attempt {attempt})...") | |
| commit_info = api.create_commit( | |
| repo_id=DEST_REPO, | |
| repo_type=REPO_TYPE, | |
| operations=operations, | |
| commit_message=f"{commit_prefix} (Batch {batch_idx}/{total_batches})", | |
| ) | |
| print(f"[OK] Batch {batch_idx}/{total_batches} committed: {commit_info.commit_url}") | |
| time.sleep(2) | |
| return True | |
| except Exception as e: | |
| print(f"[!] Batch {batch_idx} attempt {attempt} failed: {e}") | |
| if attempt < max_retries: | |
| time.sleep(5 * attempt) | |
| else: | |
| raise e | |
| def main(): | |
| api = HfApi() | |
| print(f"[*] Preparing operations for destination repo: {DEST_REPO}") | |
| operations = [] | |
| # 1. NCERT Science Files (Hindi & English) | |
| ncert_repo = "oss-codes/NCERT-Conversational-Dataset-Indic" | |
| ncert_files = api.list_repo_files(ncert_repo, repo_type="dataset") | |
| science_keywords = [ | |
| "physics", "chemistry", "biology", "science - vi", "science - vii", | |
| "science - viii", "science - ix", "science-x" | |
| ] | |
| selected_ncert = [] | |
| for f in ncert_files: | |
| if f.startswith("hi/") or f.startswith("en/"): | |
| f_lower = f.lower() | |
| if any(k in f_lower for k in science_keywords): | |
| if "political" in f_lower or "social" in f_lower or "human ecology" in f_lower: | |
| continue | |
| selected_ncert.append(f) | |
| print(f"[*] Selected {len(selected_ncert)} NCERT Science files (Hindi & English).") | |
| for f in selected_ncert: | |
| lang = "hindi" if f.startswith("hi/") else "english" | |
| base_name = f.split("/")[-1].replace("_sharegpt_conversations.jsonl", ".jsonl").replace(" ", "_").lower() | |
| dest_path = f"science/ncert_{lang}_{base_name}" | |
| operations.append( | |
| CommitOperationCopy( | |
| src_repo_id=ncert_repo, | |
| src_path_in_repo=f, | |
| path_in_repo=dest_path, | |
| src_repo_type="dataset", | |
| ) | |
| ) | |
| # 2. SciQ Question Answering & Reasoning (Physics, Chemistry, Biology) | |
| sciq_repo = "allenai/sciq" | |
| sciq_files = [ | |
| ("data/train-00000-of-00001.parquet", "science/sciq_train.parquet"), | |
| ("data/validation-00000-of-00001.parquet", "science/sciq_validation.parquet"), | |
| ("data/test-00000-of-00001.parquet", "science/sciq_test.parquet"), | |
| ] | |
| for src_file, dest_file in sciq_files: | |
| operations.append( | |
| CommitOperationCopy( | |
| src_repo_id=sciq_repo, | |
| src_path_in_repo=src_file, | |
| path_in_repo=dest_file, | |
| src_repo_type="dataset", | |
| ) | |
| ) | |
| # 3. RedMod Pure Science Textbooks (30 Parquet Shards, 11.85 GB) | |
| redmod_repo = "RedMod/science_textbooks" | |
| for i in range(30): | |
| src_file = f"part-{i:05d}.parquet" | |
| dest_file = f"science/textbooks_part_{i:05d}.parquet" | |
| operations.append( | |
| CommitOperationCopy( | |
| src_repo_id=redmod_repo, | |
| src_path_in_repo=src_file, | |
| path_in_repo=dest_file, | |
| src_repo_type="dataset", | |
| ) | |
| ) | |
| print(f"[*] Total operations to execute: {len(operations)}") | |
| batches = [operations[i:i + BATCH_SIZE] for i in range(0, len(operations), BATCH_SIZE)] | |
| print(f"[*] Split into {len(batches)} batches of up to {BATCH_SIZE} operations each.") | |
| t0 = time.time() | |
| for b_idx, b_ops in enumerate(batches, start=1): | |
| commit_batch(api, b_ops, b_idx, len(batches), "Add science stream datasets") | |
| elapsed = time.time() - t0 | |
| print(f"[OK] All {len(batches)} batches successfully committed in {elapsed:.2f} seconds!") | |
| if __name__ == "__main__": | |
| main() | |