File size: 3,263 Bytes
798598c
 
 
 
 
 
 
 
51b7902
798598c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
57f2f3e
798598c
57f2f3e
 
 
798598c
 
 
 
 
 
 
 
 
 
 
 
 
 
57f2f3e
798598c
 
 
 
 
 
 
51b7902
 
 
 
 
798598c
 
 
 
 
 
 
57f2f3e
 
 
 
 
 
 
 
51b7902
798598c
 
51b7902
57f2f3e
 
798598c
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
"""
Step 4 cleaning + dedup + eval-split (Kaggle pe chalao, data ke paas).
- Input:  data/raw/hinglish.txt, hindi.txt, english.txt
- Output: data/processed/<lang>.txt (train) + data/eval/<lang>_eval.txt (1000 lines, seed-fixed)
- Rules: empty/short drop, script-mix check, exact-dedup, eval leakage 0 (eval lines train se nikaal do)
Run (repo root se):
  python data/scripts/clean_dedup.py
"""
import hashlib
import pathlib
import random
import re

DEV = re.compile(r'[\u0900-\u097F]')
LAT = re.compile(r'[A-Za-z]')

RAW = pathlib.Path("data/raw")
PROC = pathlib.Path("data/processed")
EVAL = pathlib.Path("data/eval")
EVAL_N = 1000
SEED = 42


def dev_ratio(s):
    dev = len(DEV.findall(s))
    lat = len(LAT.findall(s))
    tot = dev + lat
    return dev / max(tot, 1)


def clean(lang, line_iterator):
    out = []
    stat = {"raw": 0, "empty": 0, "short": 0, "script": 0}
    for ln in line_iterator:
        stat["raw"] += 1
        s = ln.strip()
        if not s:
            stat["empty"] += 1
            continue
        if len(s) < 10 or len(s.split()) < 3:
            stat["short"] += 1
            continue
        r = dev_ratio(s)
        if lang == "hinglish" and (not LAT.search(s) or r > 0.3):
            stat["script"] += 1
            continue
        if lang == "hindi" and not DEV.search(s):
            stat["script"] += 1
            continue
        if lang == "english" and (DEV.search(s) or not LAT.search(s)):
            stat["script"] += 1
            continue
        out.append(s)
    stat["kept"] = len(out)
    return out, stat


def stable_hash(s: str) -> int:
    # FIX: builtin hash() is PYTHONHASHSEED-randomized -> dedup unstable across runs.
    return int.from_bytes(hashlib.blake2b(s.encode("utf-8"), digest_size=8).digest(), "big")


def main():
    PROC.mkdir(parents=True, exist_ok=True)
    EVAL.mkdir(parents=True, exist_ok=True)
    rng = random.Random(SEED)
    print("=== CLEAN REPORT ===")
    for lang in ("hinglish", "hindi", "english"):
        fp = RAW / f"{lang}.txt"
        if not fp.exists():
            print(f"[warn] {fp} nahi mili — skip")
            continue
        def stream_file(p):
            with open(p, "r", encoding="utf-8") as f:
                for line in f:
                    yield line
        kept, stat = clean(lang, stream_file(fp))
        # memory-safe exact dedup via stable hash (FIX: hash() -> blake2b)
        seen, ded = set(), []
        for s in kept:
            h = stable_hash(s)
            if h not in seen:
                seen.add(h)
                ded.append(s)
        stat["dup"] = len(kept) - len(ded)
        # eval split (random, seed-fixed) — leakage 0
        rng.shuffle(ded)
        ev, tr = ded[:EVAL_N], ded[EVAL_N:]
        open(EVAL / f"{lang}_eval.txt", "w", encoding="utf-8").write("\n".join(ev) + "\n")
        open(PROC / f"{lang}.txt", "w", encoding="utf-8").write("\n".join(tr) + "\n")
        avg = sum(len(s.split()) for s in tr) // max(len(tr), 1)
        print(f"{lang}: raw={stat['raw']} empty={stat['empty']} short={stat['short']} "
              f"script_out={stat['script']} dup={stat['dup']} -> train={len(tr)} eval={len(ev)} avg_words={avg}")
    print("=== COPY TO TEST_REPORT ===")


if __name__ == "__main__":
    main()