lhc-0-brain / code /verify_v3.py
sayed125's picture
update code/verify_v3.py (post-review release)
115d68f verified
Raw History Blame Contribute Delete
3.32 kB
#!/usr/bin/env python3
# -*- coding: utf-8 -*-
"""verify_v3.py — إثبات التكافؤ التام: features_v3 مقابل الأصل، (idx,val) عنصر-بعنصر،
+ المتجهات النهائية مع الأوزان، + حالات حافة قاسية."""
import sys, time, random
sys.path.insert(0,"/home/user/lhc/code"); sys.path.insert(0,"/home/user/lhc/recon")
import numpy as np
from phase_c_encoder import TrainableEncoder, hashed_features
from phase_p2_slot import hashed_features_slot
from features_v3 import hashed_features_slot_v3, cache_stats
from phase_w1_data import download_corpus
def same(a, b):
ia, va = a; ib, vb = b
return (ia.shape == ib.shape and va.shape == vb.shape
and np.array_equal(ia, ib) and np.array_equal(va, vb))
# ---------- 1) حالات حافة ----------
edge = ["", " ", "123", "abc123def", "٤٢", "٠", "Hello, World!", "مرحبا بالعالم 2026",
"A"*300, "ك"*50, "١٢٣ ٤٥٦", "x", "#", "3.14 2,718", "TED 2020 talk",
"ﷲ", "aaa aaa aaa", "one two three four five", "رقم 42 و 42", "MiXeD CaSe"]
fails = 0
for t in edge:
if not same(hashed_features_slot(t), hashed_features_slot_v3(t)):
fails += 1; print("✗ حالة حافة:", repr(t[:40]))
print(f"حالات الحافة: {len(edge)-fails}/{len(edge)} مطابقة")
# ---------- 2) نصوص حقيقية من المجموعات الثلاث ----------
ps = []
for name,url in [("Tanzil","https://object.pouta.csc.fi/OPUS-Tanzil/v1/moses/ar-en.txt.zip"),
("Bible","https://object.pouta.csc.fi/OPUS-bible-uedin/v1/moses/ar-en.txt.zip"),
("NeuLab-TED","https://object.pouta.csc.fi/OPUS-NeuLab-TedTalks/v1/moses/ar-en.txt.zip")]:
ps += download_corpus(name,url,20000,'data/opus')
texts = [p["ar"] for p in ps[::13]] + [p["en"] for p in ps[::13]]
DEV="/home/user/lhc/data/flores101_dataset/devtest"
ld=lambda p:[l.strip() for l in open(p,encoding="utf-8") if l.strip()]
texts += ld(f"{DEV}/ara.devtest")[:400] + ld(f"{DEV}/eng.devtest")[:400]
print(f"إجمالي نصوص التحقق: {len(texts):,}")
t0=time.time(); bad=0
for i,t in enumerate(texts):
if not same(hashed_features_slot(t), hashed_features_slot_v3(t)):
bad += 1
if bad<=3: print("✗ عدم تطابق:", repr(t[:60]))
print(f"ميزات: {len(texts)-bad:,}/{len(texts):,} مطابقة عنصر-بعنصر ({time.time()-t0:.1f} ث)")
print("المخزون بعد التحقق:", cache_stats())
# ---------- 3) المتجهات النهائية مع كل الأوزان الثلاثة ----------
for tag, wp in [("W4","weights/phase_w1_W.npz"),("W4.1","weights/phase_w1_W_robust.npz")]:
z=np.load(wp)
e1=TrainableEncoder(F=int(z["F"]),d=int(z["d"]),tau=float(z["tau"]),W=z["W"],features_fn=hashed_features_slot)
e2=TrainableEncoder(F=int(z["F"]),d=int(z["d"]),tau=float(z["tau"]),W=z["W"],features_fn=hashed_features_slot_v3)
sub = texts[:600]
V1=np.stack([e1.encode(t) for t in sub]); V2=np.stack([e2.encode(t) for t in sub])
mx = float(np.abs(V1-V2).max()); eq = bool(np.array_equal(V1,V2))
print(f"المتجهات [{tag}]: متطابقة تمامًا={eq} | أقصى فرق={mx}")
assert eq, "المتجهات ليست متطابقة!"
print("✓✓ كل التحققات نجحت")