Download code/recall_planted.py from sayed125/lhc-0-brain: direct link, hf CLI and curl.
- Browser
- Download file 4.79 kB
-
https://huggingface.co/sayed125/lhc-0-brain/resolve/main/code/recall_planted.py
- Command line
-
hf download hf://sayed125/lhc-0-brain/code/recall_planted.py
-
curl -L -o recall_planted.py https://huggingface.co/sayed125/lhc-0-brain/resolve/main/code/recall_planted.py
4.79 kB
| #!/usr/bin/env python3 | |
| # -*- coding: utf-8 -*- | |
| """recall_planted.py — قياس استدعاء مولّد المرشحين (v7: مؤشر معكوس على أندر الكلمات) | |
| بمكررات مزروعة: 500 جملة تدريب + تشوهات (تنقيط/تشكيل/تطبيع/كلمة-اثنتين…)، ثم: | |
| gen_v7 : هل توجد مرساة من مراسي P داخل الجملة المصدر S؟ (هذا ما يجعل S يجلب P) | |
| gen_v6 : فلتر simhash القديم (Hamming≤4) — للمقارنة | |
| e2e : gen ∧ معيار التشابه (jaccard≥0.45 ∧ containment≥0.65) | |
| """ | |
| import json, random, hashlib, re, sys | |
| import numpy as np | |
| sys.path.insert(0,"/home/user/lhc/recon") | |
| exec(open('/home/user/lhc/recon/build_masks_stream.py').read().split('# ---------------- مجموعات التقييم')[0]) | |
| sys.path.insert(0,"/home/user/lhc/code") | |
| from phase_w1_data import download_corpus | |
| N=500; SEED=7 | |
| DMAX=60; NANCHOR=4; J_TH=0.45; CT_TH=0.65 | |
| rng=random.Random(SEED) | |
| # جمل تدريب حقيقية | |
| src=[] | |
| with open("/var/tmp/lhc_w4/pairs.jsonl", encoding="utf-8") as f: | |
| for i,line in enumerate(f): | |
| if i>=200000: break | |
| if i%397==0: | |
| o=json.loads(line) | |
| if o.get("ar") and 8<=len(o["ar"].split())<=40: src.append(o["ar"]) | |
| if len(src)>=N: break | |
| print(f"جمل التدريب المصدر: {len(src)}") | |
| # قاموس df شبيه بجانب الاستعلامات (نستخدم مصادر متاحة: نصوص المصدر نفسها كتقدير) | |
| toks_all=[tokens(norm_ar(s)) for s in src] | |
| df={} | |
| for tk in toks_all: | |
| for t in set(tk): df[t]=df.get(t,0)+1 | |
| _HARAKAT=["\u064E","\u064F","\u0650","\u0652","\u0651","\u064B","\u064C","\u064D"] | |
| def pert_punct(t,nrs): | |
| s=t | |
| for a,b in (("\"","«"),("\"","»"),(","," ، "),("."," ."),(":"," :"),(";"," ؛"),("?"," ؟"),("!"," !")): | |
| s=s.replace(a,b) | |
| if nrs.rand()<0.5: s="« "+s+" »" | |
| return s | |
| def pert_tashkeel(t,nrs): | |
| out=[] | |
| for ch in t: | |
| out.append(ch) | |
| if re.match(r"[\u0621-\u064A]",ch) and nrs.rand()<0.30: out.append(_HARAKAT[int(nrs.randint(len(_HARAKAT)))]) | |
| return "".join(out) | |
| def pert_norm(t,nrs): | |
| for a,b in (("أ","ا"),("إ","ا"),("آ","ا"),("ى","ي"),("ة","ه")): | |
| t=t.replace(a,b) | |
| return t.translate(str.maketrans("0123456789","٠١٢٣٤٥٦٧٨٩")) | |
| def pert_words(t,k,nrs): | |
| ws=t.split(" ") | |
| idxs=[i for i,w in enumerate(ws) if len(w)>3] | |
| if len(idxs)<=k: return t | |
| for i in nrs.choice(idxs, size=k, replace=False): | |
| ws[i]=rng.choice(["الجديد","أيضا","فعلا","ربما","تماما","جدا","هنا","الآن","كذلك","بشكل"]) | |
| return " ".join(ws) | |
| variants=[ | |
| ("control", lambda t,nrs: t), | |
| ("punct", pert_punct), | |
| ("tashkeel-full", pert_tashkeel), | |
| ("normalize", pert_norm), | |
| ("word-1", lambda t,nrs: pert_words(t,1,nrs)), | |
| ("word-2", lambda t,nrs: pert_words(t,2,nrs)), | |
| ("mixed", lambda t,nrs: pert_norm(pert_punct(t,nrs),nrs)), | |
| ] | |
| out={} | |
| for name,fn in variants: | |
| g7=0; g6=0; e2e=0; e2e_wide=0; no_anchor=0 | |
| for si,S in enumerate(src): | |
| nrs=np.random.RandomState(SEED+si) | |
| P=fn(S,nrs) | |
| aq=set(tokens(norm_ar(P))); sq=set(tokens(norm_ar(S))) | |
| # مراسي P حسب استراتيجية v7 (أندر الكلمات بشرط df≤DMAX، حد NANCHOR) | |
| rare=sorted(aq, key=lambda t: df.get(t,0)) | |
| anchors=[t for t in rare if df.get(t,0)<=DMAX][:NANCHOR] | |
| if not anchors: no_anchor+=1 | |
| gen7 = any(a in sq for a in anchors) | |
| # فلتر v6 القديم | |
| h=0 | |
| if aq and sq: | |
| sh_p=simhash64(tokens(norm_ar(P))); sh_s=simhash64(tokens(norm_ar(S))) | |
| h=(sh_p^sh_s).bit_count() | |
| gen6 = (h<=4) | |
| # المعيار | |
| x=len(aq&sq); j=x/(len(aq|sq)) if (aq or sq) else 0; ct=x/min(len(aq),len(sq)) if (aq and sq) else 0 | |
| strong = (j>=J_TH and ct>=CT_TH) | |
| wide = (ct>=CT_TH) | |
| g7+=int(gen7); g6+=int(gen6) | |
| e2e+=int(gen7 and strong); e2e_wide+=int(gen7 and wide) | |
| n=len(src) | |
| out[name]={"n":n,"gen_v7_pct":round(100*g7/n,1),"gen_v6_pct":round(100*g6/n,1), | |
| "e2e_strong_pct":round(100*e2e/n,1),"e2e_wide_pct":round(100*e2e_wide/n,1), | |
| "no_anchor":no_anchor} | |
| print(f"[{name:13s}] استدعاء-مولّد v7={100*g7/n:5.1f}% | v6(simhash≤4)={100*g6/n:5.1f}% | شامل(قوي)={100*e2e/n:5.1f}% | شامل(واسع)={100*e2e_wide/n:5.1f}% | بلا مرساة={no_anchor}") | |
| json.dump({"config":{"N":N,"seed":SEED,"DMAX":DMAX,"NANCHOR":NANCHOR,"J":J_TH,"CT":CT_TH}, | |
| "results":out}, open("/home/user/lhc/results/recall_planted.json","w"), ensure_ascii=False, indent=1) | |
| print("SAVED results/recall_planted.json") | |