Download code/fasttext_local_eval.py from sayed125/lhc-0-brain: direct link, hf CLI and curl.
- Browser
- Download file 7.39 kB
-
https://huggingface.co/sayed125/lhc-0-brain/resolve/main/code/fasttext_local_eval.py
- Command line
-
hf download hf://sayed125/lhc-0-brain/code/fasttext_local_eval.py
-
curl -L -o fasttext_local_eval.py https://huggingface.co/sayed125/lhc-0-brain/resolve/main/code/fasttext_local_eval.py
7.39 kB
| #!/usr/bin/env python3 | |
| # -*- coding: utf-8 -*- | |
| """fasttext_local_eval.py — إكمال نواة Kaggle (ج) محليًا وبالضبط: | |
| نفس النصوص (ar.txt/en.txt من مخرجات النواة) + نفس البذور (11 للمحاذاة، 41 للتقييم) | |
| + نفس الرياضيات (متوسط حسابي لمتجهات الكلمات ثم تطبيع L2 ثم Procrustes). | |
| بوابة تحقق: ||XW−Y|| يجب أن يطابق 0.513 المُسجَّل في النواة (وإلا نُفشل التشغيل). | |
| المخرجات: results/fasttext_procrustes_results.json""" | |
| import os, sys, re, json, time, random, subprocess | |
| import numpy as np | |
| T0 = time.time() | |
| def log(*a): print(f"[{time.time()-T0:7.1f}s]", *a, flush=True) | |
| D = "/var/tmp/lhc_c" | |
| CLI = "/var/tmp/ft/fasttext_cli" | |
| MA, ME = f"{D}/cc.ar.128.bin", f"{D}/cc.en.128.bin" | |
| TOK = re.compile(r"\w+", re.UNICODE) | |
| DIM = 128 | |
| # ---------- (1) إعادة بناء عيّنة المحاذاة بالضبط (random.Random(11)، أول 1.5M زوجًا) ---------- | |
| rng = random.Random(11) | |
| p_keep = 250000 / 1500000 | |
| al_ar, al_en = [], [] | |
| with open(f"{D}/ar.txt", encoding="utf-8") as fa, open(f"{D}/en.txt", encoding="utf-8") as fe: | |
| for i, (a, e) in enumerate(zip(fa, fe)): | |
| if i >= 1500000: break | |
| if rng.random() < p_keep: | |
| al_ar.append(a.strip()); al_en.append(e.strip()) | |
| log(f"عيّنة المحاذاة: {len(al_ar):,} زوجًا (النواة سجّلت 250,354)") | |
| # ---------- (2) جدول متجهات الكلمات عبر CLI (dedup) ---------- | |
| def word_table(model_path, texts, tag): | |
| """يبني (vocab tokens → مصفوفة متجهات) لأول ظهور لكل كلمة في النصوص.""" | |
| vocab = {} | |
| idxs = [] # list of array('i') لكل نص | |
| from array import array | |
| flat = array('i'); offs = array('i', [0]) | |
| for t in texts: | |
| for w in TOK.findall(t.lower()): | |
| j = vocab.get(w) | |
| if j is None: | |
| j = len(vocab); vocab[w] = j | |
| flat.append(j) | |
| offs.append(len(flat)) | |
| V = len(vocab) | |
| log(f" [{tag}] نصوص={len(texts):,} وحدات={len(flat):,} مفردات فريدة={V:,}") | |
| # اكتب المفردات وأطلق CLI | |
| tf = f"{D}/vocab_{tag}.txt" | |
| with open(tf, "w", encoding="utf-8") as f: | |
| f.write("\n".join(vocab.keys())) | |
| table = np.zeros((V, DIM), dtype=np.float32) | |
| p = subprocess.Popen([CLI, "print-word-vectors", model_path], | |
| stdin=open(tf, "rb"), stdout=subprocess.PIPE, text=True, encoding="utf-8", bufsize=1<<20) | |
| n = 0 | |
| for line in p.stdout: | |
| parts = line.split() # الفراغ الزائد في نهاية السطر يطلب split() العامة | |
| if len(parts) != DIM + 1: continue | |
| j = vocab.get(parts[0]) | |
| if j is None: continue | |
| table[j] = np.array(parts[1:], dtype=np.float32) | |
| n += 1 | |
| if n % 200000 == 0: log(f" {tag}: {n:,}/{V:,}") | |
| p.wait() | |
| log(f" [{tag}] تم استلام {n:,}/{V:,} متجهًا (rc={p.returncode})") | |
| assert n >= V * 0.999, f"متجهات ناقصة: {n}/{V}" | |
| return table, flat, offs | |
| def sent_vecs(table, flat, offs): | |
| N = len(offs) - 1 | |
| out = np.zeros((N, DIM), dtype=np.float32) | |
| for i in range(N): | |
| a, b = offs[i], offs[i+1] | |
| if b <= a: continue | |
| v = table[np.frombuffer(flat, dtype=np.int32, count=b-a, offset=4*a)].mean(axis=0) | |
| nv = np.linalg.norm(v) | |
| if nv > 0: v = v / nv | |
| out[i] = v | |
| return out | |
| log("بناء جدول الكلمات AR ...") | |
| tabA, flatA, offsA = word_table(MA, al_ar, "ar") | |
| XA = sent_vecs(tabA, flatA, offsA); del tabA; log(" XA جاهز") | |
| log("بناء جدول الكلمات EN ...") | |
| tabE, flatE, offsE = word_table(ME, al_en, "en") | |
| YA = sent_vecs(tabE, flatE, offsE); del tabE; log(" YA جاهز") | |
| # ---------- (3) Procrustes (نفس نواة C) ---------- | |
| M = XA.T @ YA | |
| U, S, Vt = np.linalg.svd(M.astype(np.float64), full_matrices=False) | |
| Wp = (U @ Vt).astype(np.float32) | |
| resid = float(np.mean(np.linalg.norm(XA @ Wp - YA, axis=1))) | |
| log(f"Procrustes: dim={M.shape}, ||XW−Y||={resid:.3f} (النواة سجّلت 0.513)") | |
| assert abs(resid - 0.513) < 0.01, f"resid={resid:.4f} لا يطابق النواة (0.513) — توقف" | |
| del XA, YA, M; | |
| # ---------- (4) دوال تقييم النواة نفسها ---------- | |
| def eval_model(texts_ar, texts_en): | |
| tb, fl, of = word_table(MA, texts_ar, "ear"); VA = sent_vecs(tb, fl, of); del tb | |
| tb, fl, of = word_table(ME, texts_en, "een"); VE = sent_vecs(tb, fl, of); del tb | |
| return VA @ Wp, VE | |
| def metrics(q, t): | |
| q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9) | |
| t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9) | |
| order = np.argsort(-(q @ t.T), axis=1); N = len(q) | |
| r1 = float(np.mean(order[:, 0] == np.arange(N))) | |
| r5 = float(np.mean([np.any(order[i, :5] == i) for i in range(N)])) | |
| rr = np.argmax(order == np.arange(N)[:, None], axis=1) + 1 | |
| return {"R@1": round(r1, 4), "R@5": round(r5, 4), "MRR": round(float(np.mean(1.0 / rr)), 4)} | |
| def twin_retrieval(q, t, texts_t): | |
| q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9) | |
| t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9) | |
| pred = (q @ t.T).argmax(axis=1) | |
| return round(float(np.mean([texts_t[j] == texts_t[i] for i, j in enumerate(pred)])), 4) | |
| # ---------- (5) FLORES ---------- | |
| DEV = "/home/user/lhc/data/flores101_dataset/devtest" | |
| load = lambda p: [l.strip() for l in open(p, encoding="utf-8") if l.strip()] | |
| AR, EN = load(f"{DEV}/ara.devtest"), load(f"{DEV}/eng.devtest") | |
| Ea, Ee = eval_model(AR, EN) | |
| res = {"flores": {"ar_to_en": metrics(Ea, Ee), "en_to_ar": metrics(Ee, Ea), "n": len(AR)}, | |
| "model": "fastText(skipgram,dim128,minn3,maxn5,bucket1M,ep5)+linear Procrustes (supervised, 250K pairs)", | |
| "align_pairs": len(al_ar), "procrustes_resid": round(resid, 4)} | |
| log(f"FLORES: ar→en {res['flores']['ar_to_en']} | en→ar {res['flores']['en_to_ar']}") | |
| # ---------- (6) المجموعات الذهبية (نفس بروتوكول twin-retrieval) ---------- | |
| sys.path.insert(0, "/home/user/lhc/code"); sys.path.insert(0, "/home/user/lhc/recon") | |
| from phase_w1_data import download_corpus | |
| HELDOUT = [("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000), | |
| ("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000), | |
| ("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)] | |
| corpora, allp = {}, [] | |
| for name, url, lim in HELDOUT: | |
| ps = download_corpus(name, url, lim, "/home/user/lhc/data/opus"); corpora[name] = ps; allp += ps | |
| corpora["ALL"] = allp | |
| res["goldens"] = {} | |
| for name, ps in corpora.items(): | |
| rng2 = random.Random(41); s = rng2.sample(ps, min(300, len(ps))) | |
| ars = [p["ar"] for p in s]; ens = [p["en"] for p in s] | |
| va, ve = eval_model(ars, ens) | |
| a1 = twin_retrieval(va, ve, ens); a2 = twin_retrieval(ve, va, ars) | |
| res["goldens"][name] = {"ar_to_en": a1, "en_to_ar": a2, "n": 300} | |
| log(f"[{name:12s}] ar→en {a1:.3f} en→ar {a2:.3f}") | |
| res["_elapsed_sec"] = round(time.time()-T0, 1) | |
| json.dump(res, open("/home/user/lhc/results/fasttext_procrustes_results.json","w"), ensure_ascii=False, indent=1) | |
| log("SAVED results/fasttext_procrustes_results.json ✓") | |