Download data/scripts/add_wikipedia.py from ViuAI/ViuMini-Dense-360M: direct link, hf CLI and curl.
- Browser
- Download file 2.92 kB
-
https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/add_wikipedia.py
- Command line
-
hf download hf://ViuAI/ViuMini-Dense-360M/data/scripts/add_wikipedia.py
-
curl -L -o add_wikipedia.py https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/add_wikipedia.py
2.92 kB
| """ | |
| Add Complete English Wikipedia (wikimedia/wikipedia - 20231101.en) | |
| to ViuAI/viu-mini-raw-pretrain Hub via fast Server-Side Hub Copy. | |
| Total: 41 Parquet Shards, ~10.83 GB (~5 Billion Tokens). | |
| """ | |
| import sys | |
| import time | |
| from huggingface_hub import HfApi, CommitOperationCopy | |
| sys.stdout.reconfigure(encoding='utf-8') | |
| def main(): | |
| api = HfApi() | |
| source_repo = "wikimedia/wikipedia" | |
| target_repo = "ViuAI/viu-mini-raw-pretrain" | |
| subfolder = "20231101.en" | |
| target_folder = "wikipedia" | |
| print("=== Adding Complete English Wikipedia to Pretraining Corpus ===") | |
| print(f"Source: {source_repo} ({subfolder})") | |
| print(f"Target: {target_repo}/{target_folder}/") | |
| # List all files in 20231101.en | |
| repo_files = api.list_repo_files(source_repo, repo_type="dataset") | |
| wiki_files = sorted([f for f in repo_files if f.startswith(f"{subfolder}/") and f.endswith(".parquet")]) | |
| print(f"Total English Wikipedia Parquet shards found: {len(wiki_files)}") | |
| # Check already copied files in target repo | |
| target_files = set(api.list_repo_files(target_repo, repo_type="dataset")) | |
| operations = [] | |
| skipped = 0 | |
| for src_path in wiki_files: | |
| filename = src_path.split("/")[-1] | |
| dst_path = f"{target_folder}/{filename}" | |
| if dst_path in target_files: | |
| print(f" [Already Present] {dst_path}") | |
| skipped += 1 | |
| continue | |
| operations.append( | |
| CommitOperationCopy( | |
| src_path_in_repo=src_path, | |
| path_in_repo=dst_path, | |
| src_repo_id=source_repo, | |
| src_repo_type="dataset" | |
| ) | |
| ) | |
| print(f"\nSkipped: {skipped} files | To Copy: {len(operations)} files") | |
| if not operations: | |
| print("[ok] All English Wikipedia shards are already present in target repo!") | |
| return | |
| # Commit in batches of 10 to ensure stability and smooth progress | |
| batch_size = 10 | |
| total_batches = (len(operations) + batch_size - 1) // batch_size | |
| print(f"Committing {len(operations)} server-side copies across {total_batches} batches...") | |
| for i in range(0, len(operations), batch_size): | |
| batch = operations[i:i+batch_size] | |
| batch_idx = (i // batch_size) + 1 | |
| print(f"\n[Batch {batch_idx}/{total_batches}] Committing {len(batch)} shards to Hub...") | |
| start_t = time.time() | |
| api.create_commit( | |
| repo_id=target_repo, | |
| repo_type="dataset", | |
| operations=batch, | |
| commit_message=f"feat(wikipedia): ingest English Wikipedia 20231101.en shards batch {batch_idx}/{total_batches}" | |
| ) | |
| elapsed = time.time() - start_t | |
| print(f" [ok] Batch {batch_idx}/{total_batches} committed successfully in {elapsed:.2f}s!") | |
| print("\n[SUCCESS] All 41 English Wikipedia shards ingested into ViuAI/viu-mini-raw-pretrain!") | |
| if __name__ == "__main__": | |
| main() | |