Download scripts/relab3.py from ehabnegm/s2pro-egy: direct link, hf CLI and curl.
- Browser
- Download file 4.1 kB
-
https://huggingface.co/ehabnegm/s2pro-egy/resolve/main/scripts/relab3.py
- Command line
-
hf download hf://ehabnegm/s2pro-egy/scripts/relab3.py
-
curl -L -o relab3.py https://huggingface.co/ehabnegm/s2pro-egy/resolve/main/scripts/relab3.py
4.1 kB
| #!/usr/bin/env python3 | |
| """Rebuild ALL .lab files from source manifests. ASCII-safe regex (\\u escapes). | |
| Priority: noselleel: soniox > text_raw > manifest text | |
| eqkawkab: text_raw > manifest text | |
| sadek/mosaif: transcripts.jsonl text | |
| All output stripped of tashkeel (harakat, Quranic marks, superscript alef, | |
| tatweel) using explicit codepoint escapes - no literal Arabic in this file. | |
| """ | |
| import json | |
| import re | |
| import unicodedata | |
| from pathlib import Path | |
| RAW = Path("/opt/work/data_raw") | |
| TASHKEEL_RE = re.compile( | |
| "[ؐ-ًؚ-ٰٟۖ-ۜ۟-۪ۨ-ۭـ]" | |
| ) | |
| WS_RE = re.compile(r"\s+") | |
| def strip(t: str) -> str: | |
| t = unicodedata.normalize("NFC", str(t)) | |
| t = "".join(c for c in t if unicodedata.category(c)[0] != "C") | |
| return WS_RE.sub(" ", TASHKEEL_RE.sub("", t)).strip() | |
| def load_map(path: Path, key="text"): | |
| m = {} | |
| if path.exists(): | |
| with open(path, encoding="utf-8") as f: | |
| for line in f: | |
| line = line.strip() | |
| if line: | |
| r = json.loads(line) | |
| v = r.get(key) | |
| if v and str(v).strip(): | |
| s = strip(v) | |
| if s: # never store empty | |
| m[r["id"]] = s | |
| return m | |
| # self-test before touching anything (ASCII escapes only) | |
| diac_in = "بَلْ دِي" # ba+fatha lam+sukun / dal+kasra ya | |
| plain_out = "بل دي" # bal di | |
| assert strip(diac_in) == plain_out, "strip() broken - aborting" | |
| plain = "بل دي إن قصة" | |
| assert strip(plain) == plain, "strip() destroys plain arabic - aborting" | |
| print("strip() self-test OK") | |
| nos = {} | |
| for fn in ("train.jsonl", "dev.jsonl"): | |
| nos.update(load_map(RAW / "noselleel-egyptian-tts" / fn)) | |
| nos.update(load_map(RAW / "noselleel-egyptian-tts/transcripts_diac.jsonl", "text_raw")) | |
| nos.update(load_map(RAW / "noselleel-egyptian-tts/transcripts_soniox/train.jsonl")) | |
| eqk = {} | |
| for fn in ("train.jsonl", "dev.jsonl"): | |
| eqk.update(load_map(RAW / "eqkawkab-egyptian-tts" / fn)) | |
| eqk.update(load_map(RAW / "eqkawkab-egyptian-tts/transcripts_diac.jsonl", "text_raw")) | |
| sad = load_map(RAW / "moustafa-sadek-egyptian-tts/transcripts.jsonl") | |
| mos = load_map(RAW / "mosaifside-egyptian-tts/transcripts.jsonl") | |
| print(f"maps: nos={len(nos)} eqk={len(eqk)} sad={len(sad)} mos={len(mos)}") | |
| assert len(nos) > 8700 and len(eqk) > 2000, "maps too small - aborting" | |
| BY_PREFIX = {"nos": nos, "eqk": eqk, "sad": sad, "mos": mos} | |
| ok = 0 | |
| missing = [] | |
| for root in (Path("/opt/work/data-fs"), Path("/opt/work/data-fs-val")): | |
| if not root.exists(): | |
| continue | |
| for wav in root.rglob("*.wav"): | |
| src = BY_PREFIX.get(wav.parent.name[:3]) | |
| text = (src or {}).get(wav.stem, "") | |
| if text: | |
| wav.with_suffix(".lab").write_text(text, encoding="utf-8") | |
| ok += 1 | |
| else: | |
| missing.append(str(wav)) | |
| print(f"labs written: {ok}; wavs without text: {len(missing)}") | |
| for p in missing[:5]: | |
| print(" NO-TEXT:", p) | |
| # verify | |
| bad_empty = bad_tash = 0 | |
| for root in (Path("/opt/work/data-fs"), Path("/opt/work/data-fs-val")): | |
| for lab in root.rglob("*.lab"): | |
| t = lab.read_text(encoding="utf-8") | |
| if not t.strip(): | |
| bad_empty += 1 | |
| elif TASHKEEL_RE.search(t): | |
| bad_tash += 1 | |
| print(f"verify: empty={bad_empty} with_tashkeel={bad_tash}") | |
| ref = eqk.get("1elysrns6eE_0096", "") | |
| Path("/opt/work/eval/ref_voice.txt").write_text(ref, encoding="utf-8") | |
| print("ref len:", len(ref)) | |
| for hj in Path("/opt/work/eval").glob("dev_holdout*.jsonl"): | |
| rows = [] | |
| for line in hj.read_text(encoding="utf-8").splitlines(): | |
| if line.strip(): | |
| r = json.loads(line) | |
| pref = "nos" if r["ds"].startswith("nos") else "eqk" | |
| t = BY_PREFIX[pref].get(r["id"], "") | |
| if t: | |
| r["text"] = t | |
| rows.append(r) | |
| hj.write_text("\n".join(json.dumps(r, ensure_ascii=False) for r in rows) + "\n", | |
| encoding="utf-8") | |
| print(f"{hj.name}: {len(rows)} rows") | |
| print("RELAB3_DONE") | |