lhc-0-brain / code /fasttext_local_eval.py
sayed125's picture
update code/fasttext_local_eval.py (post-review release)
b89f271 verified
Raw History Blame Contribute Delete
7.39 kB
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fasttext_local_eval.py — إكمال نواة Kaggle (ج) محليًا وبالضبط:
نفس النصوص (ar.txt/en.txt من مخرجات النواة) + نفس البذور (11 للمحاذاة، 41 للتقييم)
+ نفس الرياضيات (متوسط حسابي لمتجهات الكلمات ثم تطبيع L2 ثم Procrustes).
بوابة تحقق: ||XW−Y|| يجب أن يطابق 0.513 المُسجَّل في النواة (وإلا نُفشل التشغيل).
المخرجات: results/fasttext_procrustes_results.json"""
import os, sys, re, json, time, random, subprocess
import numpy as np
T0 = time.time()
def log(*a): print(f"[{time.time()-T0:7.1f}s]", *a, flush=True)
D = "/var/tmp/lhc_c"
CLI = "/var/tmp/ft/fasttext_cli"
MA, ME = f"{D}/cc.ar.128.bin", f"{D}/cc.en.128.bin"
TOK = re.compile(r"\w+", re.UNICODE)
DIM = 128
# ---------- (1) إعادة بناء عيّنة المحاذاة بالضبط (random.Random(11)، أول 1.5M زوجًا) ----------
rng = random.Random(11)
p_keep = 250000 / 1500000
al_ar, al_en = [], []
with open(f"{D}/ar.txt", encoding="utf-8") as fa, open(f"{D}/en.txt", encoding="utf-8") as fe:
for i, (a, e) in enumerate(zip(fa, fe)):
if i >= 1500000: break
if rng.random() < p_keep:
al_ar.append(a.strip()); al_en.append(e.strip())
log(f"عيّنة المحاذاة: {len(al_ar):,} زوجًا (النواة سجّلت 250,354)")
# ---------- (2) جدول متجهات الكلمات عبر CLI (dedup) ----------
def word_table(model_path, texts, tag):
"""يبني (vocab tokens → مصفوفة متجهات) لأول ظهور لكل كلمة في النصوص."""
vocab = {}
idxs = [] # list of array('i') لكل نص
from array import array
flat = array('i'); offs = array('i', [0])
for t in texts:
for w in TOK.findall(t.lower()):
j = vocab.get(w)
if j is None:
j = len(vocab); vocab[w] = j
flat.append(j)
offs.append(len(flat))
V = len(vocab)
log(f" [{tag}] نصوص={len(texts):,} وحدات={len(flat):,} مفردات فريدة={V:,}")
# اكتب المفردات وأطلق CLI
tf = f"{D}/vocab_{tag}.txt"
with open(tf, "w", encoding="utf-8") as f:
f.write("\n".join(vocab.keys()))
table = np.zeros((V, DIM), dtype=np.float32)
p = subprocess.Popen([CLI, "print-word-vectors", model_path],
stdin=open(tf, "rb"), stdout=subprocess.PIPE, text=True, encoding="utf-8", bufsize=1<<20)
n = 0
for line in p.stdout:
parts = line.split() # الفراغ الزائد في نهاية السطر يطلب split() العامة
if len(parts) != DIM + 1: continue
j = vocab.get(parts[0])
if j is None: continue
table[j] = np.array(parts[1:], dtype=np.float32)
n += 1
if n % 200000 == 0: log(f" {tag}: {n:,}/{V:,}")
p.wait()
log(f" [{tag}] تم استلام {n:,}/{V:,} متجهًا (rc={p.returncode})")
assert n >= V * 0.999, f"متجهات ناقصة: {n}/{V}"
return table, flat, offs
def sent_vecs(table, flat, offs):
N = len(offs) - 1
out = np.zeros((N, DIM), dtype=np.float32)
for i in range(N):
a, b = offs[i], offs[i+1]
if b <= a: continue
v = table[np.frombuffer(flat, dtype=np.int32, count=b-a, offset=4*a)].mean(axis=0)
nv = np.linalg.norm(v)
if nv > 0: v = v / nv
out[i] = v
return out
log("بناء جدول الكلمات AR ...")
tabA, flatA, offsA = word_table(MA, al_ar, "ar")
XA = sent_vecs(tabA, flatA, offsA); del tabA; log(" XA جاهز")
log("بناء جدول الكلمات EN ...")
tabE, flatE, offsE = word_table(ME, al_en, "en")
YA = sent_vecs(tabE, flatE, offsE); del tabE; log(" YA جاهز")
# ---------- (3) Procrustes (نفس نواة C) ----------
M = XA.T @ YA
U, S, Vt = np.linalg.svd(M.astype(np.float64), full_matrices=False)
Wp = (U @ Vt).astype(np.float32)
resid = float(np.mean(np.linalg.norm(XA @ Wp - YA, axis=1)))
log(f"Procrustes: dim={M.shape}, ||XW−Y||={resid:.3f} (النواة سجّلت 0.513)")
assert abs(resid - 0.513) < 0.01, f"resid={resid:.4f} لا يطابق النواة (0.513) — توقف"
del XA, YA, M;
# ---------- (4) دوال تقييم النواة نفسها ----------
def eval_model(texts_ar, texts_en):
tb, fl, of = word_table(MA, texts_ar, "ear"); VA = sent_vecs(tb, fl, of); del tb
tb, fl, of = word_table(ME, texts_en, "een"); VE = sent_vecs(tb, fl, of); del tb
return VA @ Wp, VE
def metrics(q, t):
q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9)
t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9)
order = np.argsort(-(q @ t.T), axis=1); N = len(q)
r1 = float(np.mean(order[:, 0] == np.arange(N)))
r5 = float(np.mean([np.any(order[i, :5] == i) for i in range(N)]))
rr = np.argmax(order == np.arange(N)[:, None], axis=1) + 1
return {"R@1": round(r1, 4), "R@5": round(r5, 4), "MRR": round(float(np.mean(1.0 / rr)), 4)}
def twin_retrieval(q, t, texts_t):
q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9)
t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9)
pred = (q @ t.T).argmax(axis=1)
return round(float(np.mean([texts_t[j] == texts_t[i] for i, j in enumerate(pred)])), 4)
# ---------- (5) FLORES ----------
DEV = "/home/user/lhc/data/flores101_dataset/devtest"
load = lambda p: [l.strip() for l in open(p, encoding="utf-8") if l.strip()]
AR, EN = load(f"{DEV}/ara.devtest"), load(f"{DEV}/eng.devtest")
Ea, Ee = eval_model(AR, EN)
res = {"flores": {"ar_to_en": metrics(Ea, Ee), "en_to_ar": metrics(Ee, Ea), "n": len(AR)},
"model": "fastText(skipgram,dim128,minn3,maxn5,bucket1M,ep5)+linear Procrustes (supervised, 250K pairs)",
"align_pairs": len(al_ar), "procrustes_resid": round(resid, 4)}
log(f"FLORES: ar→en {res['flores']['ar_to_en']} | en→ar {res['flores']['en_to_ar']}")
# ---------- (6) المجموعات الذهبية (نفس بروتوكول twin-retrieval) ----------
sys.path.insert(0, "/home/user/lhc/code"); sys.path.insert(0, "/home/user/lhc/recon")
from phase_w1_data import download_corpus
HELDOUT = [("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000),
("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000),
("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)]
corpora, allp = {}, []
for name, url, lim in HELDOUT:
ps = download_corpus(name, url, lim, "/home/user/lhc/data/opus"); corpora[name] = ps; allp += ps
corpora["ALL"] = allp
res["goldens"] = {}
for name, ps in corpora.items():
rng2 = random.Random(41); s = rng2.sample(ps, min(300, len(ps)))
ars = [p["ar"] for p in s]; ens = [p["en"] for p in s]
va, ve = eval_model(ars, ens)
a1 = twin_retrieval(va, ve, ens); a2 = twin_retrieval(ve, va, ars)
res["goldens"][name] = {"ar_to_en": a1, "en_to_ar": a2, "n": 300}
log(f"[{name:12s}] ar→en {a1:.3f} en→ar {a2:.3f}")
res["_elapsed_sec"] = round(time.time()-T0, 1)
json.dump(res, open("/home/user/lhc/results/fasttext_procrustes_results.json","w"), ensure_ascii=False, indent=1)
log("SAVED results/fasttext_procrustes_results.json ✓")