Download data/scripts/clean_dedup.py from ViuAI/ViuMini-Dense-360M: direct link, hf CLI and curl.
- Browser
- Download file 3.26 kB
-
https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/clean_dedup.py
- Command line
-
hf download hf://ViuAI/ViuMini-Dense-360M/data/scripts/clean_dedup.py
-
curl -L -o clean_dedup.py https://huggingface.co/ViuAI/ViuMini-Dense-360M/resolve/main/data/scripts/clean_dedup.py
3.26 kB
| """ | |
| Step 4 cleaning + dedup + eval-split (Kaggle pe chalao, data ke paas). | |
| - Input: data/raw/hinglish.txt, hindi.txt, english.txt | |
| - Output: data/processed/<lang>.txt (train) + data/eval/<lang>_eval.txt (1000 lines, seed-fixed) | |
| - Rules: empty/short drop, script-mix check, exact-dedup, eval leakage 0 (eval lines train se nikaal do) | |
| Run (repo root se): | |
| python data/scripts/clean_dedup.py | |
| """ | |
| import hashlib | |
| import pathlib | |
| import random | |
| import re | |
| DEV = re.compile(r'[\u0900-\u097F]') | |
| LAT = re.compile(r'[A-Za-z]') | |
| RAW = pathlib.Path("data/raw") | |
| PROC = pathlib.Path("data/processed") | |
| EVAL = pathlib.Path("data/eval") | |
| EVAL_N = 1000 | |
| SEED = 42 | |
| def dev_ratio(s): | |
| dev = len(DEV.findall(s)) | |
| lat = len(LAT.findall(s)) | |
| tot = dev + lat | |
| return dev / max(tot, 1) | |
| def clean(lang, line_iterator): | |
| out = [] | |
| stat = {"raw": 0, "empty": 0, "short": 0, "script": 0} | |
| for ln in line_iterator: | |
| stat["raw"] += 1 | |
| s = ln.strip() | |
| if not s: | |
| stat["empty"] += 1 | |
| continue | |
| if len(s) < 10 or len(s.split()) < 3: | |
| stat["short"] += 1 | |
| continue | |
| r = dev_ratio(s) | |
| if lang == "hinglish" and (not LAT.search(s) or r > 0.3): | |
| stat["script"] += 1 | |
| continue | |
| if lang == "hindi" and not DEV.search(s): | |
| stat["script"] += 1 | |
| continue | |
| if lang == "english" and (DEV.search(s) or not LAT.search(s)): | |
| stat["script"] += 1 | |
| continue | |
| out.append(s) | |
| stat["kept"] = len(out) | |
| return out, stat | |
| def stable_hash(s: str) -> int: | |
| # FIX: builtin hash() is PYTHONHASHSEED-randomized -> dedup unstable across runs. | |
| return int.from_bytes(hashlib.blake2b(s.encode("utf-8"), digest_size=8).digest(), "big") | |
| def main(): | |
| PROC.mkdir(parents=True, exist_ok=True) | |
| EVAL.mkdir(parents=True, exist_ok=True) | |
| rng = random.Random(SEED) | |
| print("=== CLEAN REPORT ===") | |
| for lang in ("hinglish", "hindi", "english"): | |
| fp = RAW / f"{lang}.txt" | |
| if not fp.exists(): | |
| print(f"[warn] {fp} nahi mili — skip") | |
| continue | |
| def stream_file(p): | |
| with open(p, "r", encoding="utf-8") as f: | |
| for line in f: | |
| yield line | |
| kept, stat = clean(lang, stream_file(fp)) | |
| # memory-safe exact dedup via stable hash (FIX: hash() -> blake2b) | |
| seen, ded = set(), [] | |
| for s in kept: | |
| h = stable_hash(s) | |
| if h not in seen: | |
| seen.add(h) | |
| ded.append(s) | |
| stat["dup"] = len(kept) - len(ded) | |
| # eval split (random, seed-fixed) — leakage 0 | |
| rng.shuffle(ded) | |
| ev, tr = ded[:EVAL_N], ded[EVAL_N:] | |
| open(EVAL / f"{lang}_eval.txt", "w", encoding="utf-8").write("\n".join(ev) + "\n") | |
| open(PROC / f"{lang}.txt", "w", encoding="utf-8").write("\n".join(tr) + "\n") | |
| avg = sum(len(s.split()) for s in tr) // max(len(tr), 1) | |
| print(f"{lang}: raw={stat['raw']} empty={stat['empty']} short={stat['short']} " | |
| f"script_out={stat['script']} dup={stat['dup']} -> train={len(tr)} eval={len(ev)} avg_words={avg}") | |
| print("=== COPY TO TEST_REPORT ===") | |
| if __name__ == "__main__": | |
| main() | |