File size: 7,390 Bytes
b89f271
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
 
1
2
3
4
5
6
7
8
9
10
11
12
13
14
15
16
17
18
19
20
21
22
23
24
25
26
27
28
29
30
31
32
33
34
35
36
37
38
39
40
41
42
43
44
45
46
47
48
49
50
51
52
53
54
55
56
57
58
59
60
61
62
63
64
65
66
67
68
69
70
71
72
73
74
75
76
77
78
79
80
81
82
83
84
85
86
87
88
89
90
91
92
93
94
95
96
97
98
99
100
101
102
103
104
105
106
107
108
109
110
111
112
113
114
115
116
117
118
119
120
121
122
123
124
125
126
127
128
129
130
131
132
133
134
135
136
137
138
139
140
141
142
143
144
145
146
147
148
149
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""fasttext_local_eval.py — إكمال نواة Kaggle (ج) محليًا وبالضبط:
نفس النصوص (ar.txt/en.txt من مخرجات النواة) + نفس البذور (11 للمحاذاة، 41 للتقييم)
+ نفس الرياضيات (متوسط حسابي لمتجهات الكلمات ثم تطبيع L2 ثم Procrustes).
بوابة تحقق: ||XW−Y|| يجب أن يطابق 0.513 المُسجَّل في النواة (وإلا نُفشل التشغيل).
المخرجات: results/fasttext_procrustes_results.json"""
import os, sys, re, json, time, random, subprocess
import numpy as np

T0 = time.time()
def log(*a): print(f"[{time.time()-T0:7.1f}s]", *a, flush=True)

D = "/var/tmp/lhc_c"
CLI = "/var/tmp/ft/fasttext_cli"
MA, ME = f"{D}/cc.ar.128.bin", f"{D}/cc.en.128.bin"
TOK = re.compile(r"\w+", re.UNICODE)
DIM = 128

# ---------- (1) إعادة بناء عيّنة المحاذاة بالضبط (random.Random(11)، أول 1.5M زوجًا) ----------
rng = random.Random(11)
p_keep = 250000 / 1500000
al_ar, al_en = [], []
with open(f"{D}/ar.txt", encoding="utf-8") as fa, open(f"{D}/en.txt", encoding="utf-8") as fe:
    for i, (a, e) in enumerate(zip(fa, fe)):
        if i >= 1500000: break
        if rng.random() < p_keep:
            al_ar.append(a.strip()); al_en.append(e.strip())
log(f"عيّنة المحاذاة: {len(al_ar):,} زوجًا (النواة سجّلت 250,354)")

# ---------- (2) جدول متجهات الكلمات عبر CLI (dedup) ----------
def word_table(model_path, texts, tag):
    """يبني (vocab tokens → مصفوفة متجهات) لأول ظهور لكل كلمة في النصوص."""
    vocab = {}
    idxs = []           # list of array('i') لكل نص
    from array import array
    flat = array('i'); offs = array('i', [0])
    for t in texts:
        for w in TOK.findall(t.lower()):
            j = vocab.get(w)
            if j is None:
                j = len(vocab); vocab[w] = j
            flat.append(j)
        offs.append(len(flat))
    V = len(vocab)
    log(f"  [{tag}] نصوص={len(texts):,} وحدات={len(flat):,} مفردات فريدة={V:,}")
    # اكتب المفردات وأطلق CLI
    tf = f"{D}/vocab_{tag}.txt"
    with open(tf, "w", encoding="utf-8") as f:
        f.write("\n".join(vocab.keys()))
    table = np.zeros((V, DIM), dtype=np.float32)
    p = subprocess.Popen([CLI, "print-word-vectors", model_path],
                         stdin=open(tf, "rb"), stdout=subprocess.PIPE, text=True, encoding="utf-8", bufsize=1<<20)
    n = 0
    for line in p.stdout:
        parts = line.split()   # الفراغ الزائد في نهاية السطر يطلب split() العامة
        if len(parts) != DIM + 1: continue
        j = vocab.get(parts[0])
        if j is None: continue
        table[j] = np.array(parts[1:], dtype=np.float32)
        n += 1
        if n % 200000 == 0: log(f"    {tag}: {n:,}/{V:,}")
    p.wait()
    log(f"  [{tag}] تم استلام {n:,}/{V:,} متجهًا (rc={p.returncode})")
    assert n >= V * 0.999, f"متجهات ناقصة: {n}/{V}"
    return table, flat, offs

def sent_vecs(table, flat, offs):
    N = len(offs) - 1
    out = np.zeros((N, DIM), dtype=np.float32)
    for i in range(N):
        a, b = offs[i], offs[i+1]
        if b <= a: continue
        v = table[np.frombuffer(flat, dtype=np.int32, count=b-a, offset=4*a)].mean(axis=0)
        nv = np.linalg.norm(v)
        if nv > 0: v = v / nv
        out[i] = v
    return out

log("بناء جدول الكلمات AR ...")
tabA, flatA, offsA = word_table(MA, al_ar, "ar")
XA = sent_vecs(tabA, flatA, offsA); del tabA; log("  XA جاهز")
log("بناء جدول الكلمات EN ...")
tabE, flatE, offsE = word_table(ME, al_en, "en")
YA = sent_vecs(tabE, flatE, offsE); del tabE; log("  YA جاهز")

# ---------- (3) Procrustes (نفس نواة C) ----------
M = XA.T @ YA
U, S, Vt = np.linalg.svd(M.astype(np.float64), full_matrices=False)
Wp = (U @ Vt).astype(np.float32)
resid = float(np.mean(np.linalg.norm(XA @ Wp - YA, axis=1)))
log(f"Procrustes: dim={M.shape}, ||XW−Y||={resid:.3f}  (النواة سجّلت 0.513)")
assert abs(resid - 0.513) < 0.01, f"resid={resid:.4f} لا يطابق النواة (0.513) — توقف"
del XA, YA, M; 

# ---------- (4) دوال تقييم النواة نفسها ----------
def eval_model(texts_ar, texts_en):
    tb, fl, of = word_table(MA, texts_ar, "ear"); VA = sent_vecs(tb, fl, of); del tb
    tb, fl, of = word_table(ME, texts_en, "een"); VE = sent_vecs(tb, fl, of); del tb
    return VA @ Wp, VE

def metrics(q, t):
    q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9)
    t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9)
    order = np.argsort(-(q @ t.T), axis=1); N = len(q)
    r1 = float(np.mean(order[:, 0] == np.arange(N)))
    r5 = float(np.mean([np.any(order[i, :5] == i) for i in range(N)]))
    rr = np.argmax(order == np.arange(N)[:, None], axis=1) + 1
    return {"R@1": round(r1, 4), "R@5": round(r5, 4), "MRR": round(float(np.mean(1.0 / rr)), 4)}

def twin_retrieval(q, t, texts_t):
    q = q / np.maximum(np.linalg.norm(q, axis=1, keepdims=True), 1e-9)
    t = t / np.maximum(np.linalg.norm(t, axis=1, keepdims=True), 1e-9)
    pred = (q @ t.T).argmax(axis=1)
    return round(float(np.mean([texts_t[j] == texts_t[i] for i, j in enumerate(pred)])), 4)

# ---------- (5) FLORES ----------
DEV = "/home/user/lhc/data/flores101_dataset/devtest"
load = lambda p: [l.strip() for l in open(p, encoding="utf-8") if l.strip()]
AR, EN = load(f"{DEV}/ara.devtest"), load(f"{DEV}/eng.devtest")
Ea, Ee = eval_model(AR, EN)
res = {"flores": {"ar_to_en": metrics(Ea, Ee), "en_to_ar": metrics(Ee, Ea), "n": len(AR)},
       "model": "fastText(skipgram,dim128,minn3,maxn5,bucket1M,ep5)+linear Procrustes (supervised, 250K pairs)",
       "align_pairs": len(al_ar), "procrustes_resid": round(resid, 4)}
log(f"FLORES: ar→en {res['flores']['ar_to_en']} | en→ar {res['flores']['en_to_ar']}")

# ---------- (6) المجموعات الذهبية (نفس بروتوكول twin-retrieval) ----------
sys.path.insert(0, "/home/user/lhc/code"); sys.path.insert(0, "/home/user/lhc/recon")
from phase_w1_data import download_corpus
HELDOUT = [("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip",20000),
           ("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip",20000),
           ("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip",20000)]
corpora, allp = {}, []
for name, url, lim in HELDOUT:
    ps = download_corpus(name, url, lim, "/home/user/lhc/data/opus"); corpora[name] = ps; allp += ps
corpora["ALL"] = allp
res["goldens"] = {}
for name, ps in corpora.items():
    rng2 = random.Random(41); s = rng2.sample(ps, min(300, len(ps)))
    ars = [p["ar"] for p in s]; ens = [p["en"] for p in s]
    va, ve = eval_model(ars, ens)
    a1 = twin_retrieval(va, ve, ens); a2 = twin_retrieval(ve, va, ars)
    res["goldens"][name] = {"ar_to_en": a1, "en_to_ar": a2, "n": 300}
    log(f"[{name:12s}] ar→en {a1:.3f} en→ar {a2:.3f}")

res["_elapsed_sec"] = round(time.time()-T0, 1)
json.dump(res, open("/home/user/lhc/results/fasttext_procrustes_results.json","w"), ensure_ascii=False, indent=1)
log("SAVED results/fasttext_procrustes_results.json ✓")