File size: 3,324 Bytes
a69d2b9
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
"""
Ingest Open-Source Real-World Toxicity, Hate Speech & Profanity Datasets
to `toxicity/` folder on `ViuAI/viu-mini-raw-pretrain` via server-side CommitOperationCopy.

Datasets Included:
  1. google/civil_comments (Train Parquets ~320 MB - 2M real internet toxic, obscene & abusive comments)
  2. tdavidson/hate_speech_offensive (Parquet ~1.56 MB - 24k real offensive & hate speech tweets)
  3. pankajbiswas6/prism-hinglish-hate-speech (CSV ~8 MB - Real-world Hinglish toxic & abusive comments)
  4. nikitadesai/hasoc (CSV ~4.95 MB - Official HASOC Hindi/English offensive text)
  5. Paul/hatecheck-hindi (CSV ~0.82 MB - Verified Hindi hate speech & profanity test set)
"""
import sys
import time
from huggingface_hub import HfApi, CommitOperationCopy

sys.stdout.reconfigure(encoding='utf-8')

OPEN_SOURCE_TOXICITY_ITEMS = [
    {
        "src_repo": "google/civil_comments",
        "src_path": "data/train-00000-of-00002.parquet",
        "dst_path": "toxicity/civil_comments_train_00000.parquet"
    },
    {
        "src_repo": "google/civil_comments",
        "src_path": "data/train-00001-of-00002.parquet",
        "dst_path": "toxicity/civil_comments_train_00001.parquet"
    },
    {
        "src_repo": "tdavidson/hate_speech_offensive",
        "src_path": "data/train-00000-of-00001.parquet",
        "dst_path": "toxicity/hate_speech_offensive_train.parquet"
    },
    {
        "src_repo": "pankajbiswas6/prism-hinglish-hate-speech",
        "src_path": "data/train.csv",
        "dst_path": "toxicity/prism_hinglish_hate_train.csv"
    },
    {
        "src_repo": "nikitadesai/hasoc",
        "src_path": "traindata-basic.csv",
        "dst_path": "toxicity/hasoc_hindi_offensive.csv"
    },
    {
        "src_repo": "Paul/hatecheck-hindi",
        "src_path": "test.csv",
        "dst_path": "toxicity/hatecheck_hindi.csv"
    }
]

def main():
    api = HfApi()
    target_repo = "ViuAI/viu-mini-raw-pretrain"

    print("=== Ingesting Open-Source Internet Toxicity & Gaali Datasets ===")
    print(f"Target Repo: {target_repo}/toxicity/")

    target_files = set(api.list_repo_files(target_repo, repo_type="dataset"))
    operations = []

    for item in OPEN_SOURCE_TOXICITY_ITEMS:
        dst = item["dst_path"]
        if dst in target_files:
            print(f"  [Already Present] {dst}")
            continue
        print(f"  [Queued] {item['src_repo']}/{item['src_path']} -> {dst}")
        operations.append(
            CommitOperationCopy(
                src_path_in_repo=item["src_path"],
                path_in_repo=dst,
                src_repo_id=item["src_repo"],
                src_repo_type="dataset"
            )
        )

    if not operations:
        print("[ok] All open-source toxicity files are already present in repo!")
        return

    print(f"\nCommitting {len(operations)} server-side copies to {target_repo}...")
    start_t = time.time()
    api.create_commit(
        repo_id=target_repo,
        repo_type="dataset",
        operations=operations,
        commit_message="feat(toxicity): ingest open-source Google Civil Comments, HASOC Hindi, Hinglish Prism & Davidson hate speech datasets"
    )
    elapsed = time.time() - start_t
    print(f"[SUCCESS] All open-source toxicity datasets committed in {elapsed:.2f}s!")

if __name__ == "__main__":
    main()