Download data/scripts/add_science_stream.py from ViuAI/ViuMini-Dense-360M: direct link, hf CLI and curl.
- Browser
- Download file 6 kB
-
https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/add_science_stream.py
- Command line
-
hf download hf://ViuAI/ViuMini-Dense-360M/data/scripts/add_science_stream.py
-
curl -L -o add_science_stream.py https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/add_science_stream.py
6 kB
| #!/usr/bin/env python3 | |
| """ | |
| Comprehensive Science, Physics, Chemistry, Biology Pretraining Stream: | |
| 1. RedMod/science_textbooks: 30 Parquet Shards (11.85 GB pure textbooks) via CommitOperationCopy | |
| 2. allenai/sciq: SciQ Train, Validation, and Test QA via CommitOperationCopy | |
| 3. NCERT Science Class 6-12 (Physics, Chemistry, Biology) in Hindi & English: | |
| Converted into clean parquet shards and uploaded. | |
| """ | |
| import os | |
| import json | |
| import time | |
| import pandas as pd | |
| import pyarrow as pa | |
| import pyarrow.parquet as pq | |
| from huggingface_hub import HfApi, CommitOperationCopy, CommitOperationDelete, hf_hub_download | |
| DEST_REPO = "ViuAI/viu-mini-raw-pretrain" | |
| REPO_TYPE = "dataset" | |
| def cleanup_test_files(api): | |
| print("[*] Cleaning up temporary test files...") | |
| try: | |
| ops = [ | |
| CommitOperationDelete(path_in_repo="science/test_textbooks_00.parquet"), | |
| CommitOperationDelete(path_in_repo="science/test_sciq_train.parquet"), | |
| ] | |
| api.create_commit(repo_id=DEST_REPO, repo_type=REPO_TYPE, operations=ops, commit_message="clean temp test files") | |
| print("[OK] Temp test files removed.") | |
| except Exception as e: | |
| print("[i] Cleanup note:", e) | |
| def copy_sciq(api): | |
| print("[*] Copying allenai/sciq shards...") | |
| sciq_files = [ | |
| ("data/train-00000-of-00001.parquet", "science/sciq_train.parquet"), | |
| ("data/validation-00000-of-00001.parquet", "science/sciq_validation.parquet"), | |
| ("data/test-00000-of-00001.parquet", "science/sciq_test.parquet"), | |
| ] | |
| ops = [ | |
| CommitOperationCopy( | |
| src_repo_id="allenai/sciq", | |
| src_path_in_repo=src, | |
| path_in_repo=dest, | |
| src_repo_type="dataset", | |
| ) | |
| for src, dest in sciq_files | |
| ] | |
| commit_info = api.create_commit(repo_id=DEST_REPO, repo_type=REPO_TYPE, operations=ops, commit_message="add allenai/sciq physics/chem/bio QA") | |
| print(f"[OK] SciQ copied: {commit_info.commit_url}") | |
| def copy_redmod_textbooks(api): | |
| print("[*] Copying RedMod/science_textbooks (30 shards, 11.85 GB)...") | |
| batch_size = 5 | |
| for start_idx in range(0, 30, batch_size): | |
| end_idx = min(start_idx + batch_size, 30) | |
| ops = [] | |
| for i in range(start_idx, end_idx): | |
| src_file = f"part-{i:05d}.parquet" | |
| dest_file = f"science/textbooks_part_{i:05d}.parquet" | |
| ops.append( | |
| CommitOperationCopy( | |
| src_repo_id="RedMod/science_textbooks", | |
| src_path_in_repo=src_file, | |
| path_in_repo=dest_file, | |
| src_repo_type="dataset", | |
| ) | |
| ) | |
| print(f"[*] Committing textbooks shards {start_idx} to {end_idx - 1}...") | |
| commit_info = api.create_commit( | |
| repo_id=DEST_REPO, | |
| repo_type=REPO_TYPE, | |
| operations=ops, | |
| commit_message=f"add science textbooks shards {start_idx:05d} to {end_idx - 1:05d}", | |
| ) | |
| print(f"[OK] Committed batch {start_idx//batch_size + 1}/6: {commit_info.commit_url}") | |
| time.sleep(1) | |
| def ingest_ncert_science(api): | |
| print("[*] Processing NCERT Class 6-12 Science (Physics, Chemistry, Biology)...") | |
| ncert_repo = "oss-codes/NCERT-Conversational-Dataset-Indic" | |
| all_files = api.list_repo_files(ncert_repo, repo_type="dataset") | |
| science_kw = ["physics", "chemistry", "biology", "science - vi", "science - vii", "science - viii", "science - ix", "science-x"] | |
| for lang_code, lang_name in [("hi", "hindi"), ("en", "english")]: | |
| matching = [] | |
| for f in all_files: | |
| if f.startswith(f"{lang_code}/"): | |
| fl = f.lower() | |
| if any(k in fl for k in science_kw): | |
| if "political" in fl or "social" in fl or "human ecology" in fl: | |
| continue | |
| matching.append(f) | |
| print(f"[*] Found {len(matching)} {lang_name} NCERT science files.") | |
| all_rows = [] | |
| for f in matching: | |
| try: | |
| local_path = hf_hub_download(repo_id=ncert_repo, filename=f, repo_type="dataset") | |
| with open(local_path, "r", encoding="utf-8") as fp: | |
| for line in fp: | |
| if line.strip(): | |
| data = json.loads(line) | |
| convs = data.get("conversations", []) | |
| text_parts = [] | |
| for c in convs: | |
| sender = c.get("from", "") | |
| val = c.get("value", "") | |
| text_parts.append(f"{sender.capitalize()}: {val}") | |
| full_text = "\n".join(text_parts) | |
| all_rows.append({"text": full_text, "source": f}) | |
| except Exception as e: | |
| print(f"[!] Error reading {f}: {e}") | |
| df = pd.DataFrame(all_rows) | |
| out_parquet = f"ncert_{lang_name}_science_class6_to_12.parquet" | |
| df.to_parquet(out_parquet, compression="zstd") | |
| print(f"[OK] Generated {out_parquet} ({len(df)} rows, {os.path.getsize(out_parquet) / (1024**2):.2f} MB)") | |
| api.upload_file( | |
| path_or_fileobj=out_parquet, | |
| path_in_repo=f"science/{out_parquet}", | |
| repo_id=DEST_REPO, | |
| repo_type=REPO_TYPE, | |
| commit_message=f"add NCERT {lang_name} science class 6-12 ({len(df)} conversations)", | |
| ) | |
| print(f"[OK] Uploaded science/{out_parquet} to hub!") | |
| if os.path.exists(out_parquet): | |
| os.remove(out_parquet) | |
| def main(): | |
| api = HfApi() | |
| print("[*] Starting complete Science stream ingestion...") | |
| t0 = time.time() | |
| cleanup_test_files(api) | |
| copy_sciq(api) | |
| copy_redmod_textbooks(api) | |
| ingest_ncert_science(api) | |
| elapsed = time.time() - t0 | |
| print(f"\n[ALL DONE] Complete Science stream successfully ingested in {elapsed:.2f} seconds!") | |
| if __name__ == "__main__": | |
| main() | |