File size: 3,324 Bytes
a69d2b9 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 | """
Ingest Open-Source Real-World Toxicity, Hate Speech & Profanity Datasets
to `toxicity/` folder on `ViuAI/viu-mini-raw-pretrain` via server-side CommitOperationCopy.
Datasets Included:
1. google/civil_comments (Train Parquets ~320 MB - 2M real internet toxic, obscene & abusive comments)
2. tdavidson/hate_speech_offensive (Parquet ~1.56 MB - 24k real offensive & hate speech tweets)
3. pankajbiswas6/prism-hinglish-hate-speech (CSV ~8 MB - Real-world Hinglish toxic & abusive comments)
4. nikitadesai/hasoc (CSV ~4.95 MB - Official HASOC Hindi/English offensive text)
5. Paul/hatecheck-hindi (CSV ~0.82 MB - Verified Hindi hate speech & profanity test set)
"""
import sys
import time
from huggingface_hub import HfApi, CommitOperationCopy
sys.stdout.reconfigure(encoding='utf-8')
OPEN_SOURCE_TOXICITY_ITEMS = [
{
"src_repo": "google/civil_comments",
"src_path": "data/train-00000-of-00002.parquet",
"dst_path": "toxicity/civil_comments_train_00000.parquet"
},
{
"src_repo": "google/civil_comments",
"src_path": "data/train-00001-of-00002.parquet",
"dst_path": "toxicity/civil_comments_train_00001.parquet"
},
{
"src_repo": "tdavidson/hate_speech_offensive",
"src_path": "data/train-00000-of-00001.parquet",
"dst_path": "toxicity/hate_speech_offensive_train.parquet"
},
{
"src_repo": "pankajbiswas6/prism-hinglish-hate-speech",
"src_path": "data/train.csv",
"dst_path": "toxicity/prism_hinglish_hate_train.csv"
},
{
"src_repo": "nikitadesai/hasoc",
"src_path": "traindata-basic.csv",
"dst_path": "toxicity/hasoc_hindi_offensive.csv"
},
{
"src_repo": "Paul/hatecheck-hindi",
"src_path": "test.csv",
"dst_path": "toxicity/hatecheck_hindi.csv"
}
]
def main():
api = HfApi()
target_repo = "ViuAI/viu-mini-raw-pretrain"
print("=== Ingesting Open-Source Internet Toxicity & Gaali Datasets ===")
print(f"Target Repo: {target_repo}/toxicity/")
target_files = set(api.list_repo_files(target_repo, repo_type="dataset"))
operations = []
for item in OPEN_SOURCE_TOXICITY_ITEMS:
dst = item["dst_path"]
if dst in target_files:
print(f" [Already Present] {dst}")
continue
print(f" [Queued] {item['src_repo']}/{item['src_path']} -> {dst}")
operations.append(
CommitOperationCopy(
src_path_in_repo=item["src_path"],
path_in_repo=dst,
src_repo_id=item["src_repo"],
src_repo_type="dataset"
)
)
if not operations:
print("[ok] All open-source toxicity files are already present in repo!")
return
print(f"\nCommitting {len(operations)} server-side copies to {target_repo}...")
start_t = time.time()
api.create_commit(
repo_id=target_repo,
repo_type="dataset",
operations=operations,
commit_message="feat(toxicity): ingest open-source Google Civil Comments, HASOC Hindi, Hinglish Prism & Davidson hate speech datasets"
)
elapsed = time.time() - start_t
print(f"[SUCCESS] All open-source toxicity datasets committed in {elapsed:.2f}s!")
if __name__ == "__main__":
main()
|