File size: 7,390 Bytes
b89f271 | 1 2 3 4 5 6 7 8 9 10 11 12 13 14 15 16 17 18 19 20 21 22 23 24 25 26 27 28 29 30 31 32 33 34 35 36 37 38 39 40 41 42 43 44 45 46 47 48 49 50 51 52 53 54 55 56 57 58 59 60 61 62 63 64 65 66 67 68 69 70 71 72 73 74 75 76 77 78 79 80 81 82 83 84 85 86 87 88 89 90 91 92 93 94 95 96 97 98 99 100 101 102 103 104 105 106 107 108 109 110 111 112 113 114 115 116 117 118 119 120 121 122 123 124 125 126 127 128 129 130 131 132 133 134 135 136 137 138 139 140 141 142 143 144 145 146 147 148 149 | #!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fasttext_local_eval.py — إكمال نواة Kaggle (ج) محليًا وبالضبط:
نفس النصوص (ar.txt/en.txt من مخرجات النواة) + نفس البذور (11 للمحاذاة، 41 للتقييم)
+ نفس الرياضيات (متوسط حسابي لمتجهات الكلمات ثم تطبيع L2 ثم Procrustes).
بوابة تحقق: ||XW−Y|| يجب أن يطابق 0.513 المُسجَّل في النواة (وإلا نُفشل التشغيل).
المخرجات: results/fasttext_procrustes_results.json"""
import os, sys, re, json, time, random, subprocess
import numpy as np
T0 = time.time()
def log(*a): print(f"[{time.time()-T0:7.1f}s]", *a, flush=True)
D = "/var/tmp/lhc_c"
CLI = "/var/tmp/ft/fasttext_cli"
MA, ME = f"{D}/cc.ar.128.bin", f"{D}/cc.en.128.bin"
TOK = re.compile(r"\w+", re.UNICODE)
DIM = 128
# ---------- (1) إعادة بناء عيّنة المحاذاة بالضبط (random.Random(11)، أول 1.5M زوجًا) ----------
rng = random.Random(11)
p_keep = 250000 / 1500000
al_ar, al_en = [], []
with open(f"{D}/ar.txt", encoding="utf-8") as fa, open(f"{D}/en.txt", encoding="utf-8") as fe:
for i, (a, e) in enumerate(zip(fa, fe)):
if i >= 1500000: break
if rng.random() < p_keep:
al_ar.append(a.strip()); al_en.append(e.strip())
log(f"عيّنة المحاذاة: {len(al_ar):,} زوجًا (النواة سجّلت 250,354)")
# ---------- (2) جدول متجهات الكلمات عبر CLI (dedup) ----------
def word_table(model_path, texts, tag):
"""يبني (vocab tokens → مصفوفة متجهات) لأول ظهور لكل كلمة في النصوص."""
vocab = {}
idxs = [] # list of array('i') لكل نص
from array import array
flat = array('i'); offs = array('i', [0])
for t in texts:
for w in TOK.findall(t.lower()):
j = vocab.get(w)
if j is None:
j = len(vocab); vocab[w] = j
flat.append(j)
offs.append(len(flat))
V = len(vocab)
log(f" [{tag}] نصوص={len(texts):,} وحدات={len(flat):,} مفردات فريدة={V:,}")
# اكتب المفردات وأطلق CLI
tf = f"{D}/vocab_{tag}.txt"
with open(tf, "w", encoding="utf-8") as f:
f.write("\n".join(vocab.keys()))
table = np.zeros((V, DIM), dtype=np.float32)
p = subprocess.Popen([CLI, "print-word-vectors", model_path],
stdin=open(tf, "rb"), stdout=subprocess.PIPE, text=True, encoding="utf-8", bufsize=1<<20)
n = 0
for line in p.stdout:
parts = line.split() # الفراغ الزائد في نهاية السطر يطلب split() العامة
if len(parts) != DIM + 1: continue
j = vocab.get(parts[0])
if j is None: continue
table[j] = np.array(parts[1:], dtype=np.float32)
n += 1
if n % 200000 == 0: log(f" {tag}: {n:,}/{V:,}")
p.wait()
log(f" [{tag}] تم استلام {n:,}/{V:,} متجهًا (rc={p.returncode})")
assert n >= V * 0.999, f"متجهات ناقصة: {n}/{V}"
return table, flat, offs
def sent_vecs(table, flat, offs):
N = len(offs) - 1
out = np.zeros((N, DIM), dtype=np.float32)
for i in range(N):
a, b = offs[i], offs[i+1]
if b <= a: continue
v = table[np.frombuffer(flat, dtype=np.int32, count=b-a, offset=4*a)].mean(axis=0)
nv = np.linalg.norm(v)
if nv > 0: v = v / nv
out[i] = v
return out
log("بناء جدول الكلمات AR ...")
tabA, flatA, offsA = word_table(MA, al_ar, "ar")
XA = sent_vecs(tabA, flatA, offsA); del tabA; log(" XA جاهز")
log("بناء جدول الكلمات EN ...")
tabE, flatE, offsE = word_table(ME, al_en, "en")
YA = sent_vecs(tabE, flatE, offsE); del tabE; log(" YA جاهز")
# ---------- (3) Procrustes (نفس نواة C) ----------
M = XA.T @ YA
U, S, Vt = np.linalg.svd(M.astype(np.float64), full_matrices=False)
Wp = (U @ Vt).astype(np.float32)
resid = float(np.mean(np.linalg.norm(XA @ Wp - YA, axis=1)))
log(f"Procrustes: dim={M.shape}, ||XW−Y||={resid:.3f} (النواة سجّلت 0.513)")
assert abs(resid - 0.513) < 0.01, f"resid={resid:.4f} لا يطابق النواة (0.513) — توقف"
del XA, YA, M;
# ---------- (4) دوال تقييم النواة نفسها ----------
def eval_model(texts_ar, texts_en):
tb, fl, of = word_table(MA, texts_ar, "ear"); VA = sent_vecs(tb, fl, of); del tb
tb, fl, of = word_table(ME, texts_en, "een"); VE = sent_vecs(tb, fl, of); del tb
return VA @ Wp, VE
def metrics(q, t):
q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9)
t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9)
order = np.argsort(-(q @ t.T), axis=1); N = len(q)
r1 = float(np.mean(order[:, 0] == np.arange(N)))
r5 = float(np.mean([np.any(order[i, :5] == i) for i in range(N)]))
rr = np.argmax(order == np.arange(N)[:, None], axis=1) + 1
return {"R@1": round(r1, 4), "R@5": round(r5, 4), "MRR": round(float(np.mean(1.0 / rr)), 4)}
def twin_retrieval(q, t, texts_t):
q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9)
t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9)
pred = (q @ t.T).argmax(axis=1)
return round(float(np.mean([texts_t[j] == texts_t[i] for i, j in enumerate(pred)])), 4)
# ---------- (5) FLORES ----------
DEV = "/home/user/lhc/data/flores101_dataset/devtest"
load = lambda p: [l.strip() for l in open(p, encoding="utf-8") if l.strip()]
AR, EN = load(f"{DEV}/ara.devtest"), load(f"{DEV}/eng.devtest")
Ea, Ee = eval_model(AR, EN)
res = {"flores": {"ar_to_en": metrics(Ea, Ee), "en_to_ar": metrics(Ee, Ea), "n": len(AR)},
"model": "fastText(skipgram,dim128,minn3,maxn5,bucket1M,ep5)+linear Procrustes (supervised, 250K pairs)",
"align_pairs": len(al_ar), "procrustes_resid": round(resid, 4)}
log(f"FLORES: ar→en {res['flores']['ar_to_en']} | en→ar {res['flores']['en_to_ar']}")
# ---------- (6) المجموعات الذهبية (نفس بروتوكول twin-retrieval) ----------
sys.path.insert(0, "/home/user/lhc/code"); sys.path.insert(0, "/home/user/lhc/recon")
from phase_w1_data import download_corpus
HELDOUT = [("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000),
("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000),
("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)]
corpora, allp = {}, []
for name, url, lim in HELDOUT:
ps = download_corpus(name, url, lim, "/home/user/lhc/data/opus"); corpora[name] = ps; allp += ps
corpora["ALL"] = allp
res["goldens"] = {}
for name, ps in corpora.items():
rng2 = random.Random(41); s = rng2.sample(ps, min(300, len(ps)))
ars = [p["ar"] for p in s]; ens = [p["en"] for p in s]
va, ve = eval_model(ars, ens)
a1 = twin_retrieval(va, ve, ens); a2 = twin_retrieval(ve, va, ars)
res["goldens"][name] = {"ar_to_en": a1, "en_to_ar": a2, "n": 300}
log(f"[{name:12s}] ar→en {a1:.3f} en→ar {a2:.3f}")
res["_elapsed_sec"] = round(time.time()-T0, 1)
json.dump(res, open("/home/user/lhc/results/fasttext_procrustes_results.json","w"), ensure_ascii=False, indent=1)
log("SAVED results/fasttext_procrustes_results.json ✓")
|