Download data/scripts/add_open_source_toxicity.py from ViuAI/ViuMini-MoE-242M: direct link, hf CLI and curl.
- Browser
- Download file 3.32 kB
-
https://huggingface.co/ViuAI/ViuMini-MoE-242M/resolve/main/data/scripts/add_open_source_toxicity.py
- Command line
-
hf download hf://ViuAI/ViuMini-MoE-242M/data/scripts/add_open_source_toxicity.py
-
curl -L -o add_open_source_toxicity.py https://huggingface.co/ViuAI/ViuMini-MoE-242M/resolve/main/data/scripts/add_open_source_toxicity.py
3.32 kB
| """ | |
| Ingest Open-Source Real-World Toxicity, Hate Speech & Profanity Datasets | |
| to `toxicity/` folder on `ViuAI/viu-mini-raw-pretrain` via server-side CommitOperationCopy. | |
| Datasets Included: | |
| 1. google/civil_comments (Train Parquets ~320 MB - 2M real internet toxic, obscene & abusive comments) | |
| 2. tdavidson/hate_speech_offensive (Parquet ~1.56 MB - 24k real offensive & hate speech tweets) | |
| 3. pankajbiswas6/prism-hinglish-hate-speech (CSV ~8 MB - Real-world Hinglish toxic & abusive comments) | |
| 4. nikitadesai/hasoc (CSV ~4.95 MB - Official HASOC Hindi/English offensive text) | |
| 5. Paul/hatecheck-hindi (CSV ~0.82 MB - Verified Hindi hate speech & profanity test set) | |
| """ | |
| import sys | |
| import time | |
| from huggingface_hub import HfApi, CommitOperationCopy | |
| sys.stdout.reconfigure(encoding='utf-8') | |
| OPEN_SOURCE_TOXICITY_ITEMS = [ | |
| { | |
| "src_repo": "google/civil_comments", | |
| "src_path": "data/train-00000-of-00002.parquet", | |
| "dst_path": "toxicity/civil_comments_train_00000.parquet" | |
| }, | |
| { | |
| "src_repo": "google/civil_comments", | |
| "src_path": "data/train-00001-of-00002.parquet", | |
| "dst_path": "toxicity/civil_comments_train_00001.parquet" | |
| }, | |
| { | |
| "src_repo": "tdavidson/hate_speech_offensive", | |
| "src_path": "data/train-00000-of-00001.parquet", | |
| "dst_path": "toxicity/hate_speech_offensive_train.parquet" | |
| }, | |
| { | |
| "src_repo": "pankajbiswas6/prism-hinglish-hate-speech", | |
| "src_path": "data/train.csv", | |
| "dst_path": "toxicity/prism_hinglish_hate_train.csv" | |
| }, | |
| { | |
| "src_repo": "nikitadesai/hasoc", | |
| "src_path": "traindata-basic.csv", | |
| "dst_path": "toxicity/hasoc_hindi_offensive.csv" | |
| }, | |
| { | |
| "src_repo": "Paul/hatecheck-hindi", | |
| "src_path": "test.csv", | |
| "dst_path": "toxicity/hatecheck_hindi.csv" | |
| } | |
| ] | |
| def main(): | |
| api = HfApi() | |
| target_repo = "ViuAI/viu-mini-raw-pretrain" | |
| print("=== Ingesting Open-Source Internet Toxicity & Gaali Datasets ===") | |
| print(f"Target Repo: {target_repo}/toxicity/") | |
| target_files = set(api.list_repo_files(target_repo, repo_type="dataset")) | |
| operations = [] | |
| for item in OPEN_SOURCE_TOXICITY_ITEMS: | |
| dst = item["dst_path"] | |
| if dst in target_files: | |
| print(f" [Already Present] {dst}") | |
| continue | |
| print(f" [Queued] {item['src_repo']}/{item['src_path']} -> {dst}") | |
| operations.append( | |
| CommitOperationCopy( | |
| src_path_in_repo=item["src_path"], | |
| path_in_repo=dst, | |
| src_repo_id=item["src_repo"], | |
| src_repo_type="dataset" | |
| ) | |
| ) | |
| if not operations: | |
| print("[ok] All open-source toxicity files are already present in repo!") | |
| return | |
| print(f"\nCommitting {len(operations)} server-side copies to {target_repo}...") | |
| start_t = time.time() | |
| api.create_commit( | |
| repo_id=target_repo, | |
| repo_type="dataset", | |
| operations=operations, | |
| commit_message="feat(toxicity): ingest open-source Google Civil Comments, HASOC Hindi, Hinglish Prism & Davidson hate speech datasets" | |
| ) | |
| elapsed = time.time() - start_t | |
| print(f"[SUCCESS] All open-source toxicity datasets committed in {elapsed:.2f}s!") | |
| if __name__ == "__main__": | |
| main() | |