""" Ingest Open-Source Real-World Toxicity, Hate Speech & Profanity Datasets to `toxicity/` folder on `ViuAI/viu-mini-raw-pretrain` via server-side CommitOperationCopy. Datasets Included: 1. google/civil_comments (Train Parquets ~320 MB - 2M real internet toxic, obscene & abusive comments) 2. tdavidson/hate_speech_offensive (Parquet ~1.56 MB - 24k real offensive & hate speech tweets) 3. pankajbiswas6/prism-hinglish-hate-speech (CSV ~8 MB - Real-world Hinglish toxic & abusive comments) 4. nikitadesai/hasoc (CSV ~4.95 MB - Official HASOC Hindi/English offensive text) 5. Paul/hatecheck-hindi (CSV ~0.82 MB - Verified Hindi hate speech & profanity test set) """ import sys import time from huggingface_hub import HfApi, CommitOperationCopy sys.stdout.reconfigure(encoding='utf-8') OPEN_SOURCE_TOXICITY_ITEMS = [ { "src_repo": "google/civil_comments", "src_path": "data/train-00000-of-00002.parquet", "dst_path": "toxicity/civil_comments_train_00000.parquet" }, { "src_repo": "google/civil_comments", "src_path": "data/train-00001-of-00002.parquet", "dst_path": "toxicity/civil_comments_train_00001.parquet" }, { "src_repo": "tdavidson/hate_speech_offensive", "src_path": "data/train-00000-of-00001.parquet", "dst_path": "toxicity/hate_speech_offensive_train.parquet" }, { "src_repo": "pankajbiswas6/prism-hinglish-hate-speech", "src_path": "data/train.csv", "dst_path": "toxicity/prism_hinglish_hate_train.csv" }, { "src_repo": "nikitadesai/hasoc", "src_path": "traindata-basic.csv", "dst_path": "toxicity/hasoc_hindi_offensive.csv" }, { "src_repo": "Paul/hatecheck-hindi", "src_path": "test.csv", "dst_path": "toxicity/hatecheck_hindi.csv" } ] def main(): api = HfApi() target_repo = "ViuAI/viu-mini-raw-pretrain" print("=== Ingesting Open-Source Internet Toxicity & Gaali Datasets ===") print(f"Target Repo: {target_repo}/toxicity/") target_files = set(api.list_repo_files(target_repo, repo_type="dataset")) operations = [] for item in OPEN_SOURCE_TOXICITY_ITEMS: dst = item["dst_path"] if dst in target_files: print(f" [Already Present] {dst}") continue print(f" [Queued] {item['src_repo']}/{item['src_path']} -> {dst}") operations.append( CommitOperationCopy( src_path_in_repo=item["src_path"], path_in_repo=dst, src_repo_id=item["src_repo"], src_repo_type="dataset" ) ) if not operations: print("[ok] All open-source toxicity files are already present in repo!") return print(f"\nCommitting {len(operations)} server-side copies to {target_repo}...") start_t = time.time() api.create_commit( repo_id=target_repo, repo_type="dataset", operations=operations, commit_message="feat(toxicity): ingest open-source Google Civil Comments, HASOC Hindi, Hinglish Prism & Davidson hate speech datasets" ) elapsed = time.time() - start_t print(f"[SUCCESS] All open-source toxicity datasets committed in {elapsed:.2f}s!") if __name__ == "__main__": main()